salverz/llm-document-parser
0
1# config.py2from pydantic import BaseModel3from datetime import date4from typing import List5 6# Options: "rapid", "easy", "ocrmac", "tesseract"7OCR_MODEL = "easy"8 9# Must be set when using the tesseract OCR model10# Linux: "/usr/share/tesseract-ocr/4.00/tessdata"11# Windows: "C:\\Program Files\\Tesseract-OCR\\tessdata"12# Mac: "/usr/local/share/tessdata" or "/opt/homebrew/share/tessdata"13TESSERACT_TESSDATA_LOCATION = "/usr/share/tesseract-ocr/4.00/tessdata"14 15OLLAMA_MODEL = "llama3:instruct"16 17LLM_PROMPT = """18 Extract all transactions from the following statement. Each transaction must be returned as a JSON object with the fields: transaction_date (YYYY-MM-DD), description, amount, and transaction_type ('deposit' or 'withdrawal'). All of these must be returned as a list of JSON objects under a key called 'transactions'. Here is an example:19 [20 {21 transaction_date: 2025-01-24,22 description: "Walmart",23 amount: 34.24,24 transaction_type: "withdrawl"25 }26 ]27"""28 29# Options: "csv", "json", "excel"30EXPORT_TYPE = "json"31 32# Can be a file or directory33INPUT_PATH = ""34OUTPUT_FOLDER = ""35OUTPUT_FILE_NAME = "output"36 37# Define Pydantic response models for instructor:38 39class BankStatementEntry(BaseModel):40 transaction_date: date | None | str41 description: str | None42 amount: float | None43 #transaction_type: Literal['deposit', 'withdrawal', None]44 transaction_type: str | None45 46class BankStatement(BaseModel):47 transactions: List[BankStatementEntry] | None48 49# The model that LLM output will conform to50RESPONSE_MODEL = BankStatement51 