Team Ai
Apppublic

vieveksharmaa/multi-source-rag

sourceHugging Faceupdated 4mo agoView on Hugging Face
0likes
debug_test.py166 linesDownload Raw Back to root
1"""2RAG Pipeline Diagnostic Script3================================4Run: python debug_test.py5 6Paste a YouTube URL when prompted.7This will test each step independently and show exactly where things break.8"""9import os, sys, shutil10sys.path.insert(0, os.path.dirname(os.path.abspath(__file__)))11os.chdir(os.path.dirname(os.path.abspath(__file__)))  # ensure correct working dir12 13from dotenv import load_dotenv14load_dotenv()15 16LINE = "─" * 6017 18def header(title):19    print(f"\n{LINE}\n  {title}\n{LINE}")20 21# ── Step 0: Config ────────────────────────────────────────────22header("STEP 0 — Config")23import config24print(f"  LLM provider   : {config.LLM_PROVIDER} / {config.GROQ_MODEL}")25print(f"  Embedding      : {config.EMBEDDING_PROVIDER} / {config.HF_EMBEDDING_MODEL}")26print(f"  Vector store   : {config.VECTOR_STORE}")27print(f"  Chroma path    : {os.path.abspath(config.CHROMA_PERSIST_DIR)}")28print(f"  Chroma exists  : {os.path.exists(config.CHROMA_PERSIST_DIR)}")29 30# ── Step 1: Wipe ChromaDB ─────────────────────────────────────31header("STEP 1 — Wipe ChromaDB (fresh start)")32db_path = os.path.abspath(config.CHROMA_PERSIST_DIR)33if os.path.exists(db_path):34    shutil.rmtree(db_path)35    print(f"  ✓ Deleted {db_path}")36else:37    print(f"  (already empty)")38 39# ── Step 2: YouTube transcript ────────────────────────────────40header("STEP 2 — YouTube Transcript API")41url = input("\n  Paste a YouTube URL: ").strip()42if not url:43    url = "https://www.youtube.com/watch?v=dQw4w9WgXcQ"44 45video_id = None46import re47for pat in [r"(?:v=|\/)([0-9A-Za-z_-]{11})", r"youtu\.be\/([0-9A-Za-z_-]{11})"]:48    m = re.search(pat, url)49    if m:50        video_id = m.group(1)51        break52 53print(f"  Video ID: {video_id}")54 55try:56    from youtube_transcript_api import YouTubeTranscriptApi57    api = YouTubeTranscriptApi()58    transcript = api.fetch(video_id)59    snippets = list(transcript)60    print(f"  ✓ Fetched {len(snippets)} transcript snippets")61    print(f"  First snippet: \"{snippets[0].text[:80]}\"  @ {snippets[0].start:.1f}s")62    print(f"  Last snippet:  \"{snippets[-1].text[:80]}\"  @ {snippets[-1].start:.1f}s")63    total_words = sum(len(s.text.split()) for s in snippets)64    print(f"  Total words: ~{total_words}")65except Exception as e:66    print(f"  ✗ TRANSCRIPT ERROR: {e}")67    sys.exit(1)68 69# ── Step 3: Chunk the transcript ──────────────────────────────70header("STEP 3 — Chunking")71from src.sources.youtube import load_youtube72try:73    docs = load_youtube(url)74    print(f"  ✓ Created {len(docs)} chunks")75    if docs:76        print(f"  First chunk ({len(docs[0].page_content)} chars):")77        print(f"    \"{docs[0].page_content[:150]}\"")78        print(f"  Metadata: {docs[0].metadata}")79except Exception as e:80    print(f"  ✗ CHUNKING ERROR: {e}")81    import traceback; traceback.print_exc()82    sys.exit(1)83 84# ── Step 4: Embeddings ────────────────────────────────────────85header("STEP 4 — Embeddings")86try:87    from src.llm.provider import get_embeddings88    embeddings = get_embeddings()89    test_vec = embeddings.embed_query("test")90    print(f"  ✓ Embedding model loaded, dimension={len(test_vec)}")91except Exception as e:92    print(f"  ✗ EMBEDDING ERROR: {e}")93    import traceback; traceback.print_exc()94    sys.exit(1)95 96# ── Step 5: Store in ChromaDB ─────────────────────────────────97header("STEP 5 — Store in ChromaDB")98try:99    from src.vectorstore.store import add_documents, get_vector_store100    ids = add_documents(docs)101    print(f"  ✓ Stored {len(ids)} chunks in ChromaDB")102    store = get_vector_store()103    if hasattr(store, '_collection'):104        total = store._collection.count()105        print(f"  ChromaDB total docs now: {total}")106except Exception as e:107    print(f"  ✗ STORE ERROR: {e}")108    import traceback; traceback.print_exc()109    sys.exit(1)110 111# ── Step 6: Retrieval ─────────────────────────────────────────112header("STEP 6 — Retrieval (similarity search)")113test_queries = [114    "What is this video about?",115    "What are the main topics discussed?",116    "Summarize the key points",117]118try:119    from src.vectorstore.store import similarity_search_with_score120    for q in test_queries:121        results = similarity_search_with_score(q, k=3)122        print(f"\n  Query: \"{q}\"")123        for doc, score in results:124            src = doc.metadata.get("source_type", "?")125            ts  = doc.metadata.get("timestamp", "")126            preview = doc.page_content[:80].replace("\n", " ")127            print(f"    score={score:.3f} [{src}] {ts}  \"{preview}\"")128except Exception as e:129    print(f"  ✗ RETRIEVAL ERROR: {e}")130    import traceback; traceback.print_exc()131    sys.exit(1)132 133# ── Step 7: LLM ───────────────────────────────────────────────134header("STEP 7 — LLM (single call test)")135try:136    from src.llm.provider import get_llm137    from langchain_core.messages import HumanMessage138    llm = get_llm()139    resp = llm.invoke([HumanMessage(content="Reply with exactly: OK")])140    print(f"  ✓ LLM responded: \"{resp.content.strip()}\"")141except Exception as e:142    print(f"  ✗ LLM ERROR: {e}")143    import traceback; traceback.print_exc()144 145# ── Step 8: Full query ────────────────────────────────────────146header("STEP 8 — Full RAG Query")147try:148    from src.graph.rag_graph import query, _compiled_graph149    import src.graph.rag_graph as rag150    rag._compiled_graph = None  # force rebuild151 152    result = query("What is this video about? Give a brief summary.")153    print(f"\n  ANSWER:\n{result['answer']}\n")154    print(f"  CITATIONS ({len(result['citations'])}):")155    for c in result['citations']:156        print(f"    [{c['number']}] {c['text']}")157except Exception as e:158    print(f"  ✗ QUERY ERROR: {e}")159    import traceback; traceback.print_exc()160 161header("DIAGNOSTIC COMPLETE")162print("  If Step 6 showed score < 0.3 for all results, the embedding")163print("  model may need warming up — try running again.")164print("  If Step 5 stored 0 chunks, the YouTube loader failed.")165print()166