Team Ai
Apppublic

MD2204/multi_modality

sourceHugging Faceupdated 8mo agoView on Hugging Face
1likes
config.py66 linesDownload Raw Back to src
1import os
2from pathlib import Path
3from dotenv import load_dotenv
4
5# Load environment variables
6load_dotenv()
7
8# Base Paths
9BASE_DIR = Path(__file__).parent.parent
10DATA_DIR = BASE_DIR / "data"
11INPUTS_DIR = DATA_DIR / "inputs"
12IMAGES_DIR = DATA_DIR / "extracted_images"
13
14# Ensure directories exist
15INPUTS_DIR.mkdir(parents=True, exist_ok=True)
16IMAGES_DIR.mkdir(parents=True, exist_ok=True)
17
18# Ollama Settings
19OLLAMA_API_KEY = os.getenv("OLLAMA_API_KEY")
20OLLAMA_MODEL = os.getenv("OLLAMA_MODEL", "gpt-oss:120b")
21OLLAMA_BASE_URL = os.getenv("OLLAMA_BASE_URL", "https://ollama.com")
22OLLAMA_TEMPERATURE = float(os.getenv("OLLAMA_TEMPERATURE", "0.1"))
23
24# Google Settings
25GOOGLE_API_KEY = os.getenv("GOOGLE_API_KEY")
26GOOGLE_MODEL = os.getenv("GOOGLE_MODEL", "gemini-2.5-flash")
27GOOGLE_TEMPERATURE = float(os.getenv("GOOGLE_TEMPERATURE", "0.1"))
28if not GOOGLE_API_KEY:
29    print("⚠️ Warning: GOOGLE_API_KEY not found in environment variables!")
30
31# Pinecone Settings
32PINECONE_API_KEY = os.getenv("PINECONE_API_KEY")
33PINECONE_INDEX_NAME = os.getenv("PINECONE_INDEX_NAME", "multimodal-assistant")
34if not PINECONE_API_KEY:
35    print("⚠️ Warning: PINECONE_API_KEY not found in environment variables!")
36
37# Model Settings
38EMBEDDING_MODEL_NAME = "sentence-transformers/all-MiniLM-L6-v2"
39EMBEDDING_DEVICE = "cpu"
40
41# text splitter settings
42CHUNK_SIZE = 700
43CHUNK_OVERLAP = 50
44
45# retrieval settings
46RETRIEVER_K = 5
47
48# Multimodal Prompts
49GEMINI_RAG_PROMPT = """
50Act as a high-precision data extraction engine for a RAG system. Your goal is to provide a dense, structured representation of the image for a downstream LLM.
51
521. **Classification**: Briefly categorize the image (e.g., Line Chart, Architecture Diagram, Logic Flow, Photo).
53
542. **Core Data Extraction**:
55   - **If a Graph/Diagram**: Define the axes, labels, and scale. Provide a Markdown 'Data Table' of the MOST CRITICAL points (start, end, peaks, valleys, and major intervals). **CRITICAL: Limit tables to max 15 key rows. Do not generate empty rows.**
56   - **If a Table/Document**: Extract the structure into Markdown. Maintain headers and column alignment.
57   - **If an Infographic/Object**: List the primary subjects, branding, and their spatial layout.
58
593. **Quantitative Precision**: Use specific numerical estimates from the visual scale. Avoid vague terms like 'significant increase'; use 'increase from ~200 to ~800'.
60
614. **Nuance & Metadata**: Identify legends, units of measurement, and any small-print technical notes.
62
635. **Concise Summary**: Provide exactly one sentence that captures the "main message" of this visual to make easier to understand it.
64
65"""
66