Team Ai
Apppublic

Ameer1606/PostgresPro-Support

sourceHugging Facemitupdated 4mo agoView on Hugging Face
0likes
ingest.py144 linesDownload Raw Back to src
1import os2import sys3from pathlib import Path4sys.path.append(str(Path(__file__).resolve().parent.parent))5from pypdf import PdfReader6from langchain_text_splitters import RecursiveCharacterTextSplitter7from langchain_community.embeddings import HuggingFaceEmbeddings8from qdrant_client import QdrantClient9from qdrant_client.http import models as rest10from src.config import DATA_DIR, QDRANT_API_KEY, QDRANT_URL, QDRANT_COLLECTION_NAME, EMBEDDING_MODEL11 12def generate_mock_files():13    """Generates mock files if they don't exist in the data directory."""14    DATA_DIR.mkdir(parents=True, exist_ok=True)15    16    manifest_path = DATA_DIR / "company_manifest.md"17    if not manifest_path.exists():18        with open(manifest_path, "w") as f:19            f.write("""# PostgresPro Support Services Manifest20 21## Contact Information22- **Phone:** +1-800-PG-DATA23- **Email:** support@postgrespro.local24 25## Operating Hours26- 24/7/365 for Critical issues (Severity 1)27- Monday-Friday, 9 AM - 5 PM EST for Standard issues (Severity 2-4)28 29## Support Escalation Paths301. Level 1: Basic troubleshooting and log collection.312. Level 2: Advanced diagnostics and configuration tuning.323. Level 3: Core internals and bug reports.33""")34        print(f"Created mock file: {manifest_path}")35 36    sla_path = DATA_DIR / "support_sla.pdf"37    if not sla_path.exists():38        # creating a simple text file with pdf extension just to satisfy constraints if needed,39        # but realistically pypdf might fail. A safer approach is to use a proper PDF generator.40        # But for mock purposes, let's just make it a readable text if it's missing, or we can write a tiny valid PDF.41        from reportlab.pdfgen import canvas42        try:43            c = canvas.Canvas(str(sla_path))44            c.drawString(100, 750, "Support Service Level Agreement (SLA)")45            c.drawString(100, 730, "Severity 1: 15 minutes response time.")46            c.drawString(100, 710, "Severity 2: 2 hours response time.")47            c.drawString(100, 690, "Severity 3: 1 business day response time.")48            c.save()49            print(f"Created mock file: {sla_path}")50        except ImportError:51            print("reportlab not installed, skipping support_sla.pdf generation.")52 53def ingest_data():54    """Reads documents, chunks them, and uploads to Qdrant."""55 56    print("Starting ingestion pipeline...")57    generate_mock_files()58 59    docs = []60    61    # Process PDF files62    for pdf_file in DATA_DIR.glob("*.pdf"):63        print(f"Processing PDF: {pdf_file.name}")64        try:65            reader = PdfReader(pdf_file)66            for i, page in enumerate(reader.pages):67                text = page.extract_text()68                if text:69                    docs.append({70                        "text": text,71                        "metadata": {"source": pdf_file.name, "page": i + 1}72                    })73        except Exception as e:74            print(f"Failed to read {pdf_file.name}: {e}")75 76    # Process MD files77    for md_file in DATA_DIR.glob("*.md"):78        print(f"Processing Markdown: {md_file.name}")79        try:80            with open(md_file, "r") as f:81                text = f.read()82                docs.append({83                    "text": text,84                    "metadata": {"source": md_file.name, "page": 1}85                })86        except Exception as e:87            print(f"Failed to read {md_file.name}: {e}")88 89    if not docs:90        print("No documents found to ingest.")91        return92 93    # Text Splitting94    text_splitter = RecursiveCharacterTextSplitter(chunk_size=1000, chunk_overlap=200)95    split_docs = []96    for doc in docs:97        chunks = text_splitter.split_text(doc["text"])98        for chunk in chunks:99            split_docs.append({100                "text": chunk,101                "metadata": doc["metadata"]102            })103    104    print(f"Split into {len(split_docs)} chunks.")105 106    # Initialize Embeddings107    embeddings = HuggingFaceEmbeddings(model_name=EMBEDDING_MODEL)108    109    # Connect to Qdrant110    try:111        client = QdrantClient(url=QDRANT_URL, api_key=QDRANT_API_KEY)112        113        # Always drop and recreate the collection so re-running ingest114        # does not produce duplicate chunks.115        collections = [c.name for c in client.get_collections().collections]116        if QDRANT_COLLECTION_NAME in collections:117            print(f"Dropping existing collection: {QDRANT_COLLECTION_NAME}")118            client.delete_collection(QDRANT_COLLECTION_NAME)119        120        # Prepare data for Qdrant121        texts = [doc["text"] for doc in split_docs]122        metadatas = [doc["metadata"] for doc in split_docs]123        124        # Embed and upload125        print("Embedding and uploading to Qdrant...")126        from langchain_qdrant import QdrantVectorStore127        128        QdrantVectorStore.from_texts(129            texts=texts,130            embedding=embeddings,131            metadatas=metadatas,132            url=QDRANT_URL,133            api_key=QDRANT_API_KEY,134            collection_name=QDRANT_COLLECTION_NAME,135        )136        print("Ingestion completed successfully.")137        138    except Exception as e:139        print(f"Error during Qdrant upload: {e}")140 141 142if __name__ == "__main__":143    ingest_data()144