Shashiguduri/github-code-explainer
0
1"""2backend/services/qa_engine.py3-------------------------------4Orchestrates the full RAG pipeline for one Q&A session.5 6Pipeline:7 1. Detect question intent (structure, architecture, dependency, etc.)8 2. Adjust retrieval strategy per intent9 3. embed_query(question) → query vector10 4. vector_store.search(vec, top_k) → top-K relevant chunks11 5. build_context(chunks) + scenario hint → grounded prompt context12 6. groq_client.ask/stream(question, ctx) → LLM answer13"""14 15import re16import logging17from dataclasses import dataclass, field18from services.embeddings import CodeEmbedder19from services.vector_store import VectorStore20from llm.groq_client import GroqLLMClient21 22logger = logging.getLogger(__name__)23 24MAX_CONTEXT_CHARS = 8_00025DEFAULT_TOP_K = 526 27 28# ══════════════════════════════════════════════════════════════════════════════29# Intent Detection30# ══════════════════════════════════════════════════════════════════════════════31 32_STRUCTURE_KEYWORDS = [33 "file structure", "folder structure", "directory structure",34 "project structure", "codebase structure", "list files",35 "all files", "file tree", "folder tree", "what files",36 "which files", "list all", "show files", "show all files",37 "project layout", "repo structure",38]39 40_ARCHITECTURE_KEYWORDS = [41 "how does the project work", "architecture", "overview",42 "how is the project organized", "high level", "high-level",43 "explain the codebase", "how does this work", "what does this project do",44 "how does the app work", "how is this built", "tech stack",45 "design pattern", "main components", "how are things connected",46 "entry point", "main flow", "data flow",47]48 49_DEPENDENCY_KEYWORDS = [50 "import", "imports", "depend", "dependency", "dependencies",51 "what does it use", "which libraries", "packages",52 "require", "requirements", "installed packages",53]54 55_DEBUGGING_KEYWORDS = [56 "bug", "error", "issue", "problem", "fix", "wrong",57 "doesn't work", "does not work", "broken", "crash",58 "exception", "fail", "failing", "why does", "why is",59 "what's wrong", "what is wrong", "debug",60]61 62_HOWTO_KEYWORDS = [63 "how to", "how do i", "how can i", "steps to",64 "guide", "tutorial", "setup", "install", "configure",65 "run the", "start the", "deploy",66]67 68_COMPARISON_KEYWORDS = [69 "difference between", "compare", "vs", "versus",70 "which is better", "pros and cons", "similar to",71]72 73# Conversational / non-code messages that should NOT trigger RAG retrieval74# These are checked as EXACT full-message matches (after normalization)75_CONVERSATIONAL_EXACT = {76 "thank you", "thanks", "thankyou", "thx", "thank you so much",77 "thanks a lot", "thanks for the help", "thanks for explaining",78 "hello", "hi", "hey", "hey there", "hi there",79 "good morning", "good evening", "good afternoon",80 "bye", "goodbye", "see you", "take care",81 "ok", "okay", "got it", "understood", "makes sense",82 "great", "awesome", "perfect", "nice", "cool", "wonderful",83 "yes", "no", "yep", "nope", "sure", "alright",84 "who are you", "what are you", "what can you do",85 "help", "lol", "haha", "hmm", "wow",86}87 88 89def _is_conversational(question: str) -> bool:90 """Return True if the message is casual/conversational, not a code question."""91 q = question.lower().strip().rstrip("?!.")92 93 # Exact match against known conversational phrases94 if q in _CONVERSATIONAL_EXACT:95 return True96 97 return False98 99 100class _QuestionIntent:101 """Detected intent with retrieval tuning parameters."""102 STRUCTURE = "structure"103 ARCHITECTURE = "architecture"104 DEPENDENCY = "dependency"105 DEBUGGING = "debugging"106 HOWTO = "howto"107 COMPARISON = "comparison"108 SPECIFIC_FILE = "specific_file"109 CONVERSATIONAL = "conversational"110 GENERAL = "general"111 112 113# Scenario-specific hints appended to the context so the LLM114# knows *how* to frame its answer.115_SCENARIO_HINTS = {116 _QuestionIntent.ARCHITECTURE: (117 "\n\n[INSTRUCTION: This is an architecture/overview question. "118 "Provide a high-level explanation of how the components connect. "119 "Start with the entry point, describe the main modules, and explain the data flow. "120 "Use a structured breakdown with headers for each component.]\n"121 ),122 _QuestionIntent.DEPENDENCY: (123 "\n\n[INSTRUCTION: This is a dependency/import question. "124 "List all imports and external libraries found in the context. "125 "Group them by: standard library, third-party packages, and internal modules. "126 "Explain what each dependency is used for.]\n"127 ),128 _QuestionIntent.DEBUGGING: (129 "\n\n[INSTRUCTION: This is a debugging/error question. "130 "Analyze the code for potential issues. Look for: unhandled edge cases, "131 "missing error handling, type mismatches, race conditions, and incorrect logic. "132 "Suggest specific fixes with code examples where possible.]\n"133 ),134 _QuestionIntent.HOWTO: (135 "\n\n[INSTRUCTION: This is a how-to/setup question. "136 "Provide step-by-step instructions. Number each step clearly. "137 "Include exact commands, file paths, and configuration values where visible in the context.]\n"138 ),139 _QuestionIntent.COMPARISON: (140 "\n\n[INSTRUCTION: This is a comparison question. "141 "Create a clear side-by-side comparison. Use a structured format with: "142 "purpose, implementation approach, pros/cons, and when to use each.]\n"143 ),144 _QuestionIntent.SPECIFIC_FILE: (145 "\n\n[INSTRUCTION: The user is asking about a specific file. "146 "Focus your explanation on that file's purpose, its exports/public API, "147 "how it connects to the rest of the codebase, and any notable patterns or edge cases.]\n"148 ),149}150 151# Top-K overrides per intent (broader questions need more context)152_INTENT_TOP_K = {153 _QuestionIntent.ARCHITECTURE: 10,154 _QuestionIntent.DEPENDENCY: 8,155 _QuestionIntent.DEBUGGING: 6,156 _QuestionIntent.HOWTO: 6,157 _QuestionIntent.COMPARISON: 8,158 _QuestionIntent.SPECIFIC_FILE: 4,159 _QuestionIntent.GENERAL: 5,160}161 162# File extensions pattern for detecting specific file references163_FILE_PATTERN = re.compile(164 r'[\w/\\]+\.(?:py|ts|tsx|js|jsx|go|rs|java|rb|css|html|json|yaml|yml|toml|md)\b',165 re.IGNORECASE,166)167 168 169def _detect_intent(question: str) -> str:170 """Classify the question to determine the best retrieval strategy."""171 q = question.lower()172 173 # Conversational check FIRST — before any code intent matching174 if _is_conversational(question):175 return _QuestionIntent.CONVERSATIONAL176 177 if any(kw in q for kw in _STRUCTURE_KEYWORDS):178 return _QuestionIntent.STRUCTURE179 180 if any(kw in q for kw in _ARCHITECTURE_KEYWORDS):181 return _QuestionIntent.ARCHITECTURE182 183 if any(kw in q for kw in _COMPARISON_KEYWORDS):184 return _QuestionIntent.COMPARISON185 186 if any(kw in q for kw in _DEBUGGING_KEYWORDS):187 return _QuestionIntent.DEBUGGING188 189 if any(kw in q for kw in _HOWTO_KEYWORDS):190 return _QuestionIntent.HOWTO191 192 if any(kw in q for kw in _DEPENDENCY_KEYWORDS):193 return _QuestionIntent.DEPENDENCY194 195 # Check if user mentioned a specific file name/path196 if _FILE_PATTERN.search(question):197 return _QuestionIntent.SPECIFIC_FILE198 199 return _QuestionIntent.GENERAL200 201 202# ══════════════════════════════════════════════════════════════════════════════203# QA Result204# ══════════════════════════════════════════════════════════════════════════════205 206@dataclass207class QAResult:208 """Structured answer returned by QAEngine.answer()."""209 answer: str210 sources: list[str] = field(default_factory=list)211 chunks_used: int = 0212 top_results: list[dict] = field(default_factory=list)213 model_used: str = ""214 intent: str = ""215 216 217# ══════════════════════════════════════════════════════════════════════════════218# QA Engine219# ══════════════════════════════════════════════════════════════════════════════220 221class QAEngine:222 """223 Coordinates embedding, retrieval, and LLM generation.224 225 One QAEngine instance per indexed repository — stored in FastAPI's226 in-memory repo_store dict.227 """228 229 def __init__(230 self,231 vector_store: VectorStore,232 embedder: CodeEmbedder,233 llm_client: GroqLLMClient,234 top_k: int = DEFAULT_TOP_K,235 ):236 self.vector_store = vector_store237 self.embedder = embedder238 self.llm_client = llm_client239 self.top_k = top_k240 241 @property242 def all_sources(self) -> list[str]:243 """Return a sorted list of every unique file path in the index."""244 seen: set[str] = set()245 for doc in self.vector_store.documents:246 src = doc.metadata.get("source", "")247 if src:248 seen.add(src)249 return sorted(seen)250 251 # ── Main answer method ───────────────────────────────────────────────252 253 def answer(self, question: str) -> QAResult:254 """255 Answer a natural-language question about the indexed repository.256 257 Detects intent → adjusts retrieval → assembles context → calls LLM.258 """259 if not self.vector_store.is_ready:260 raise RuntimeError("Vector store is not ready. Index the repository first.")261 if not question.strip():262 raise ValueError("Question cannot be empty.")263 264 intent = _detect_intent(question)265 logger.info("Question intent: %s | %s", intent, question[:100])266 267 # ── Conversational: no RAG needed ─────────────────────────────────268 if intent == _QuestionIntent.CONVERSATIONAL:269 return self._handle_conversational(question)270 271 # ── Structure questions: bypass vector search ────────────────────272 if intent == _QuestionIntent.STRUCTURE:273 return self._handle_structure(question)274 275 # ── Architecture questions: inject file list + broader retrieval ─276 if intent == _QuestionIntent.ARCHITECTURE:277 return self._handle_architecture(question)278 279 # ── All other intents: standard RAG with tuned top_k ─────────────280 top_k = _INTENT_TOP_K.get(intent, self.top_k)281 query_vec = self.embedder.embed_query(question)282 results = self.vector_store.search(query_vec, top_k=top_k)283 logger.info("Retrieved %d chunks (top_k=%d).", len(results), top_k)284 285 if not results:286 return QAResult(287 answer=(288 "I could not find relevant code for your question. "289 "Try rephrasing or ensure the repository was indexed correctly."290 ),291 intent=intent,292 )293 294 context, sources, chunks_used = _build_context(results, MAX_CONTEXT_CHARS)295 296 # Append scenario-specific hint if available297 hint = _SCENARIO_HINTS.get(intent, "")298 context += hint299 300 answer_text = self.llm_client.ask(question=question, context=context)301 302 return QAResult(303 answer=answer_text,304 sources=sources,305 chunks_used=chunks_used,306 top_results=results,307 model_used=self.llm_client.model,308 intent=intent,309 )310 311 # ── Streaming answer method ──────────────────────────────────────────312 313 def stream_answer(self, question: str):314 """315 Stream the LLM response token-by-token for a given question.316 317 Yields JSON-serialisable dicts:318 {"token": "..."} — one per LLM output chunk319 {"done": True, "sources": [...], — final metadata packet320 "chunks_used": int, "model": str, "intent": str}321 """322 if not self.vector_store.is_ready:323 raise RuntimeError("Vector store is not ready. Index the repository first.")324 if not question.strip():325 raise ValueError("Question cannot be empty.")326 327 intent = _detect_intent(question)328 logger.info("[stream] Intent: %s | %s", intent, question[:100])329 330 # ── Conversational: instant reply ─────────────────────────────────331 if intent == _QuestionIntent.CONVERSATIONAL:332 yield from self._stream_conversational(question)333 return334 335 # ── Structure questions ──────────────────────────────────────────336 if intent == _QuestionIntent.STRUCTURE:337 yield from self._stream_structure(question)338 return339 340 # ── Architecture questions ───────────────────────────────────────341 if intent == _QuestionIntent.ARCHITECTURE:342 yield from self._stream_architecture(question)343 return344 345 # ── All other intents ────────────────────────────────────────────346 top_k = _INTENT_TOP_K.get(intent, self.top_k)347 query_vec = self.embedder.embed_query(question)348 results = self.vector_store.search(query_vec, top_k=top_k)349 350 if not results:351 yield {"token": "I could not find relevant code for your question. Try rephrasing or ask about a different part of the codebase."}352 yield {"done": True, "sources": [], "chunks_used": 0,353 "model": self.llm_client.model, "intent": intent}354 return355 356 context, sources, chunks_used = _build_context(results, MAX_CONTEXT_CHARS)357 hint = _SCENARIO_HINTS.get(intent, "")358 context += hint359 360 logger.info("[stream] Retrieved %d chunks (top_k=%d). Streaming ...", chunks_used, top_k)361 362 for token in self.llm_client.stream_tokens(question=question, context=context):363 yield {"token": token}364 365 yield {366 "done": True,367 "sources": sources,368 "chunks_used": chunks_used,369 "model": self.llm_client.model,370 "intent": intent,371 }372 373 # ══════════════════════════════════════════════════════════════════════374 # Specialised handlers375 # ══════════════════════════════════════════════════════════════════════376 377 def _handle_conversational(self, question: str) -> QAResult:378 """Handle greetings, thanks, and casual messages without RAG."""379 response = _get_conversational_response(question)380 logger.info("Conversational message — responding without RAG.")381 return QAResult(382 answer=response,383 sources=[],384 chunks_used=0,385 model_used="none",386 intent=_QuestionIntent.CONVERSATIONAL,387 )388 389 def _stream_conversational(self, question: str):390 """Streaming variant for conversational messages."""391 response = _get_conversational_response(question)392 logger.info("[stream] Conversational message — instant reply.")393 yield {"token": response}394 yield {"done": True, "sources": [], "chunks_used": 0,395 "model": "none", "intent": _QuestionIntent.CONVERSATIONAL}396 397 def _handle_structure(self, question: str) -> QAResult:398 """Full file list injection — bypasses vector search entirely."""399 sources = self.all_sources400 context = (401 "The following is the complete list of all files indexed "402 "from this repository (one per line):\n\n"403 + "\n".join(sources)404 + "\n\n[INSTRUCTION: Present the file structure as a clean, "405 "indented directory tree. Group files by their directories. "406 "Add a brief one-line description for key files if their purpose "407 "is obvious from the name.]\n"408 )409 logger.info("Structure question — injecting %d file paths.", len(sources))410 answer_text = self.llm_client.ask(question=question, context=context)411 return QAResult(412 answer=answer_text,413 sources=sources[:20],414 chunks_used=len(sources),415 model_used=self.llm_client.model,416 intent=_QuestionIntent.STRUCTURE,417 )418 419 def _handle_architecture(self, question: str) -> QAResult:420 """Combine file list overview + broader chunk retrieval for architecture."""421 sources = self.all_sources422 file_overview = (423 "Complete file list for architectural context:\n"424 + "\n".join(sources) + "\n\n"425 )426 427 # Also retrieve semantic chunks for deeper context428 top_k = _INTENT_TOP_K[_QuestionIntent.ARCHITECTURE]429 query_vec = self.embedder.embed_query(question)430 results = self.vector_store.search(query_vec, top_k=top_k)431 432 chunk_context, chunk_sources, chunks_used = _build_context(433 results, MAX_CONTEXT_CHARS - len(file_overview)434 )435 436 context = (437 file_overview + chunk_context438 + _SCENARIO_HINTS[_QuestionIntent.ARCHITECTURE]439 )440 441 all_sources = list(dict.fromkeys(sources[:10] + chunk_sources))442 answer_text = self.llm_client.ask(question=question, context=context)443 444 return QAResult(445 answer=answer_text,446 sources=all_sources,447 chunks_used=chunks_used + len(sources),448 top_results=results,449 model_used=self.llm_client.model,450 intent=_QuestionIntent.ARCHITECTURE,451 )452 453 def _stream_structure(self, question: str):454 """Streaming variant of _handle_structure."""455 sources = self.all_sources456 context = (457 "The following is the complete list of all files indexed "458 "from this repository (one per line):\n\n"459 + "\n".join(sources)460 + "\n\n[INSTRUCTION: Present the file structure as a clean, "461 "indented directory tree. Group files by directories.]\n"462 )463 logger.info("[stream] Structure question — %d file paths.", len(sources))464 for token in self.llm_client.stream_tokens(question=question, context=context):465 yield {"token": token}466 yield {"done": True, "sources": sources[:20], "chunks_used": len(sources),467 "model": self.llm_client.model, "intent": _QuestionIntent.STRUCTURE}468 469 def _stream_architecture(self, question: str):470 """Streaming variant of _handle_architecture."""471 sources = self.all_sources472 file_overview = (473 "Complete file list for architectural context:\n"474 + "\n".join(sources) + "\n\n"475 )476 top_k = _INTENT_TOP_K[_QuestionIntent.ARCHITECTURE]477 query_vec = self.embedder.embed_query(question)478 results = self.vector_store.search(query_vec, top_k=top_k)479 480 chunk_context, chunk_sources, chunks_used = _build_context(481 results, MAX_CONTEXT_CHARS - len(file_overview)482 )483 context = (484 file_overview + chunk_context485 + _SCENARIO_HINTS[_QuestionIntent.ARCHITECTURE]486 )487 all_sources = list(dict.fromkeys(sources[:10] + chunk_sources))488 489 logger.info("[stream] Architecture question — %d paths + %d chunks.", len(sources), chunks_used)490 for token in self.llm_client.stream_tokens(question=question, context=context):491 yield {"token": token}492 yield {"done": True, "sources": all_sources,493 "chunks_used": chunks_used + len(sources),494 "model": self.llm_client.model, "intent": _QuestionIntent.ARCHITECTURE}495 496 497# ══════════════════════════════════════════════════════════════════════════════498# Conversational Response Helper499# ══════════════════════════════════════════════════════════════════════════════500 501def _get_conversational_response(question: str) -> str:502 """Return a friendly response for casual/conversational messages."""503 q = question.lower().strip().rstrip("?!.")504 505 # Thank you variants506 if any(w in q for w in ["thank", "thanks", "thankyou", "thx"]):507 return (508 "You're welcome! 😊 Feel free to ask anything else "509 "about the codebase — I'm here to help!"510 )511 512 # Greetings513 if any(w in q for w in ["hello", "hi", "hey", "good morning", "good evening", "good afternoon"]):514 return (515 "Hey there! 👋 I'm ready to help you explore this repository. "516 "Ask me anything about the code — how it works, what a file does, "517 "the project structure, or any specific question!"518 )519 520 # Farewells521 if any(w in q for w in ["bye", "goodbye", "see you", "take care"]):522 return "Goodbye! 👋 Happy coding, and feel free to come back anytime!"523 524 # Acknowledgements525 if any(w in q for w in ["ok", "okay", "got it", "understood", "makes sense"]):526 return (527 "Great! Let me know if you have any more questions "528 "about the codebase. 🚀"529 )530 531 # Positive feedback532 if any(w in q for w in ["great", "awesome", "perfect", "nice", "cool", "wonderful"]):533 return "Glad I could help! 😄 Ask away if you need anything else!"534 535 # Identity questions536 if any(w in q for w in ["who are you", "what are you", "what can you do"]):537 return (538 "I'm a **Code Explainer AI** 🤖 — I analyze GitHub repositories and answer "539 "your questions about the code. You can ask me about:\n\n"540 "- 📂 **File structure** — \"Show the project layout\"\n"541 "- 🏗️ **Architecture** — \"How does this project work?\"\n"542 "- 🔍 **Specific code** — \"What does auth.py do?\"\n"543 "- 🐛 **Debugging** — \"Why might this break?\"\n"544 "- 📦 **Dependencies** — \"What libraries does this use?\"\n\n"545 "Just ask in plain English!"546 )547 548 # Generic fallback for other conversational messages549 return (550 "I'm here to help you understand the code! "551 "Ask me any question about the repository. 😊"552 )553 554 555# ══════════════════════════════════════════════════════════════════════════════556# Context Builder557# ══════════════════════════════════════════════════════════════════════════════558 559def _build_context(560 results: list[dict],561 max_chars: int,562) -> tuple[str, list[str], int]:563 """564 Concatenate retrieved chunks into a single context string.565 Truncates to max_chars to stay inside the LLM context window.566 """567 parts: list[str] = []568 sources: list[str] = []569 total = 0570 used = 0571 572 for r in results:573 source = r["source"]574 score = r["score"]575 text = r["document"].page_content576 header = f"# File: {source} (relevance: {score:.2f})\n"577 block = header + text + "\n\n---\n\n"578 579 if total + len(block) > max_chars:580 remaining = max_chars - total - len(header) - 10581 if remaining > 100:582 parts.append(header + text[:remaining] + "\n… [truncated]\n\n---\n\n")583 used += 1584 break585 586 parts.append(block)587 total += len(block)588 used += 1589 if source not in sources:590 sources.append(source)591 592 return "".join(parts), sources, used593 