Team Ai
Apppublic

Shashiguduri/github-code-explainer

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
qa_engine.py593 linesDownload Raw Back to services
1"""2backend/services/qa_engine.py3-------------------------------4Orchestrates the full RAG pipeline for one Q&A session.5 6Pipeline:7  1. Detect question intent (structure, architecture, dependency, etc.)8  2. Adjust retrieval strategy per intent9  3. embed_query(question) → query vector10  4. vector_store.search(vec, top_k) → top-K relevant chunks11  5. build_context(chunks) + scenario hint → grounded prompt context12  6. groq_client.ask/stream(question, ctx) → LLM answer13"""14 15import re16import logging17from dataclasses import dataclass, field18from services.embeddings import CodeEmbedder19from services.vector_store import VectorStore20from llm.groq_client import GroqLLMClient21 22logger = logging.getLogger(__name__)23 24MAX_CONTEXT_CHARS = 8_00025DEFAULT_TOP_K     = 526 27 28# ══════════════════════════════════════════════════════════════════════════════29# Intent Detection30# ══════════════════════════════════════════════════════════════════════════════31 32_STRUCTURE_KEYWORDS = [33    "file structure", "folder structure", "directory structure",34    "project structure", "codebase structure", "list files",35    "all files", "file tree", "folder tree", "what files",36    "which files", "list all", "show files", "show all files",37    "project layout", "repo structure",38]39 40_ARCHITECTURE_KEYWORDS = [41    "how does the project work", "architecture", "overview",42    "how is the project organized", "high level", "high-level",43    "explain the codebase", "how does this work", "what does this project do",44    "how does the app work", "how is this built", "tech stack",45    "design pattern", "main components", "how are things connected",46    "entry point", "main flow", "data flow",47]48 49_DEPENDENCY_KEYWORDS = [50    "import", "imports", "depend", "dependency", "dependencies",51    "what does it use", "which libraries", "packages",52    "require", "requirements", "installed packages",53]54 55_DEBUGGING_KEYWORDS = [56    "bug", "error", "issue", "problem", "fix", "wrong",57    "doesn't work", "does not work", "broken", "crash",58    "exception", "fail", "failing", "why does", "why is",59    "what's wrong", "what is wrong", "debug",60]61 62_HOWTO_KEYWORDS = [63    "how to", "how do i", "how can i", "steps to",64    "guide", "tutorial", "setup", "install", "configure",65    "run the", "start the", "deploy",66]67 68_COMPARISON_KEYWORDS = [69    "difference between", "compare", "vs", "versus",70    "which is better", "pros and cons", "similar to",71]72 73# Conversational / non-code messages that should NOT trigger RAG retrieval74# These are checked as EXACT full-message matches (after normalization)75_CONVERSATIONAL_EXACT = {76    "thank you", "thanks", "thankyou", "thx", "thank you so much",77    "thanks a lot", "thanks for the help", "thanks for explaining",78    "hello", "hi", "hey", "hey there", "hi there",79    "good morning", "good evening", "good afternoon",80    "bye", "goodbye", "see you", "take care",81    "ok", "okay", "got it", "understood", "makes sense",82    "great", "awesome", "perfect", "nice", "cool", "wonderful",83    "yes", "no", "yep", "nope", "sure", "alright",84    "who are you", "what are you", "what can you do",85    "help", "lol", "haha", "hmm", "wow",86}87 88 89def _is_conversational(question: str) -> bool:90    """Return True if the message is casual/conversational, not a code question."""91    q = question.lower().strip().rstrip("?!.")92 93    # Exact match against known conversational phrases94    if q in _CONVERSATIONAL_EXACT:95        return True96 97    return False98 99 100class _QuestionIntent:101    """Detected intent with retrieval tuning parameters."""102    STRUCTURE     = "structure"103    ARCHITECTURE  = "architecture"104    DEPENDENCY    = "dependency"105    DEBUGGING     = "debugging"106    HOWTO         = "howto"107    COMPARISON    = "comparison"108    SPECIFIC_FILE = "specific_file"109    CONVERSATIONAL = "conversational"110    GENERAL       = "general"111 112 113# Scenario-specific hints appended to the context so the LLM114# knows *how* to frame its answer.115_SCENARIO_HINTS = {116    _QuestionIntent.ARCHITECTURE: (117        "\n\n[INSTRUCTION: This is an architecture/overview question. "118        "Provide a high-level explanation of how the components connect. "119        "Start with the entry point, describe the main modules, and explain the data flow. "120        "Use a structured breakdown with headers for each component.]\n"121    ),122    _QuestionIntent.DEPENDENCY: (123        "\n\n[INSTRUCTION: This is a dependency/import question. "124        "List all imports and external libraries found in the context. "125        "Group them by: standard library, third-party packages, and internal modules. "126        "Explain what each dependency is used for.]\n"127    ),128    _QuestionIntent.DEBUGGING: (129        "\n\n[INSTRUCTION: This is a debugging/error question. "130        "Analyze the code for potential issues. Look for: unhandled edge cases, "131        "missing error handling, type mismatches, race conditions, and incorrect logic. "132        "Suggest specific fixes with code examples where possible.]\n"133    ),134    _QuestionIntent.HOWTO: (135        "\n\n[INSTRUCTION: This is a how-to/setup question. "136        "Provide step-by-step instructions. Number each step clearly. "137        "Include exact commands, file paths, and configuration values where visible in the context.]\n"138    ),139    _QuestionIntent.COMPARISON: (140        "\n\n[INSTRUCTION: This is a comparison question. "141        "Create a clear side-by-side comparison. Use a structured format with: "142        "purpose, implementation approach, pros/cons, and when to use each.]\n"143    ),144    _QuestionIntent.SPECIFIC_FILE: (145        "\n\n[INSTRUCTION: The user is asking about a specific file. "146        "Focus your explanation on that file's purpose, its exports/public API, "147        "how it connects to the rest of the codebase, and any notable patterns or edge cases.]\n"148    ),149}150 151# Top-K overrides per intent (broader questions need more context)152_INTENT_TOP_K = {153    _QuestionIntent.ARCHITECTURE: 10,154    _QuestionIntent.DEPENDENCY:   8,155    _QuestionIntent.DEBUGGING:    6,156    _QuestionIntent.HOWTO:        6,157    _QuestionIntent.COMPARISON:   8,158    _QuestionIntent.SPECIFIC_FILE: 4,159    _QuestionIntent.GENERAL:      5,160}161 162# File extensions pattern for detecting specific file references163_FILE_PATTERN = re.compile(164    r'[\w/\\]+\.(?:py|ts|tsx|js|jsx|go|rs|java|rb|css|html|json|yaml|yml|toml|md)\b',165    re.IGNORECASE,166)167 168 169def _detect_intent(question: str) -> str:170    """Classify the question to determine the best retrieval strategy."""171    q = question.lower()172 173    # Conversational check FIRST — before any code intent matching174    if _is_conversational(question):175        return _QuestionIntent.CONVERSATIONAL176 177    if any(kw in q for kw in _STRUCTURE_KEYWORDS):178        return _QuestionIntent.STRUCTURE179 180    if any(kw in q for kw in _ARCHITECTURE_KEYWORDS):181        return _QuestionIntent.ARCHITECTURE182 183    if any(kw in q for kw in _COMPARISON_KEYWORDS):184        return _QuestionIntent.COMPARISON185 186    if any(kw in q for kw in _DEBUGGING_KEYWORDS):187        return _QuestionIntent.DEBUGGING188 189    if any(kw in q for kw in _HOWTO_KEYWORDS):190        return _QuestionIntent.HOWTO191 192    if any(kw in q for kw in _DEPENDENCY_KEYWORDS):193        return _QuestionIntent.DEPENDENCY194 195    # Check if user mentioned a specific file name/path196    if _FILE_PATTERN.search(question):197        return _QuestionIntent.SPECIFIC_FILE198 199    return _QuestionIntent.GENERAL200 201 202# ══════════════════════════════════════════════════════════════════════════════203# QA Result204# ══════════════════════════════════════════════════════════════════════════════205 206@dataclass207class QAResult:208    """Structured answer returned by QAEngine.answer()."""209    answer:      str210    sources:     list[str]  = field(default_factory=list)211    chunks_used: int        = 0212    top_results: list[dict] = field(default_factory=list)213    model_used:  str        = ""214    intent:      str        = ""215 216 217# ══════════════════════════════════════════════════════════════════════════════218# QA Engine219# ══════════════════════════════════════════════════════════════════════════════220 221class QAEngine:222    """223    Coordinates embedding, retrieval, and LLM generation.224 225    One QAEngine instance per indexed repository — stored in FastAPI's226    in-memory repo_store dict.227    """228 229    def __init__(230        self,231        vector_store: VectorStore,232        embedder: CodeEmbedder,233        llm_client: GroqLLMClient,234        top_k: int = DEFAULT_TOP_K,235    ):236        self.vector_store = vector_store237        self.embedder     = embedder238        self.llm_client   = llm_client239        self.top_k        = top_k240 241    @property242    def all_sources(self) -> list[str]:243        """Return a sorted list of every unique file path in the index."""244        seen: set[str] = set()245        for doc in self.vector_store.documents:246            src = doc.metadata.get("source", "")247            if src:248                seen.add(src)249        return sorted(seen)250 251    # ── Main answer method ───────────────────────────────────────────────252 253    def answer(self, question: str) -> QAResult:254        """255        Answer a natural-language question about the indexed repository.256 257        Detects intent → adjusts retrieval → assembles context → calls LLM.258        """259        if not self.vector_store.is_ready:260            raise RuntimeError("Vector store is not ready. Index the repository first.")261        if not question.strip():262            raise ValueError("Question cannot be empty.")263 264        intent = _detect_intent(question)265        logger.info("Question intent: %s | %s", intent, question[:100])266 267        # ── Conversational: no RAG needed ─────────────────────────────────268        if intent == _QuestionIntent.CONVERSATIONAL:269            return self._handle_conversational(question)270 271        # ── Structure questions: bypass vector search ────────────────────272        if intent == _QuestionIntent.STRUCTURE:273            return self._handle_structure(question)274 275        # ── Architecture questions: inject file list + broader retrieval ─276        if intent == _QuestionIntent.ARCHITECTURE:277            return self._handle_architecture(question)278 279        # ── All other intents: standard RAG with tuned top_k ─────────────280        top_k = _INTENT_TOP_K.get(intent, self.top_k)281        query_vec = self.embedder.embed_query(question)282        results = self.vector_store.search(query_vec, top_k=top_k)283        logger.info("Retrieved %d chunks (top_k=%d).", len(results), top_k)284 285        if not results:286            return QAResult(287                answer=(288                    "I could not find relevant code for your question. "289                    "Try rephrasing or ensure the repository was indexed correctly."290                ),291                intent=intent,292            )293 294        context, sources, chunks_used = _build_context(results, MAX_CONTEXT_CHARS)295 296        # Append scenario-specific hint if available297        hint = _SCENARIO_HINTS.get(intent, "")298        context += hint299 300        answer_text = self.llm_client.ask(question=question, context=context)301 302        return QAResult(303            answer=answer_text,304            sources=sources,305            chunks_used=chunks_used,306            top_results=results,307            model_used=self.llm_client.model,308            intent=intent,309        )310 311    # ── Streaming answer method ──────────────────────────────────────────312 313    def stream_answer(self, question: str):314        """315        Stream the LLM response token-by-token for a given question.316 317        Yields JSON-serialisable dicts:318          {"token": "..."}                  — one per LLM output chunk319          {"done": True, "sources": [...],  — final metadata packet320           "chunks_used": int, "model": str, "intent": str}321        """322        if not self.vector_store.is_ready:323            raise RuntimeError("Vector store is not ready. Index the repository first.")324        if not question.strip():325            raise ValueError("Question cannot be empty.")326 327        intent = _detect_intent(question)328        logger.info("[stream] Intent: %s | %s", intent, question[:100])329 330        # ── Conversational: instant reply ─────────────────────────────────331        if intent == _QuestionIntent.CONVERSATIONAL:332            yield from self._stream_conversational(question)333            return334 335        # ── Structure questions ──────────────────────────────────────────336        if intent == _QuestionIntent.STRUCTURE:337            yield from self._stream_structure(question)338            return339 340        # ── Architecture questions ───────────────────────────────────────341        if intent == _QuestionIntent.ARCHITECTURE:342            yield from self._stream_architecture(question)343            return344 345        # ── All other intents ────────────────────────────────────────────346        top_k = _INTENT_TOP_K.get(intent, self.top_k)347        query_vec = self.embedder.embed_query(question)348        results = self.vector_store.search(query_vec, top_k=top_k)349 350        if not results:351            yield {"token": "I could not find relevant code for your question. Try rephrasing or ask about a different part of the codebase."}352            yield {"done": True, "sources": [], "chunks_used": 0,353                   "model": self.llm_client.model, "intent": intent}354            return355 356        context, sources, chunks_used = _build_context(results, MAX_CONTEXT_CHARS)357        hint = _SCENARIO_HINTS.get(intent, "")358        context += hint359 360        logger.info("[stream] Retrieved %d chunks (top_k=%d). Streaming ...", chunks_used, top_k)361 362        for token in self.llm_client.stream_tokens(question=question, context=context):363            yield {"token": token}364 365        yield {366            "done":        True,367            "sources":     sources,368            "chunks_used": chunks_used,369            "model":       self.llm_client.model,370            "intent":      intent,371        }372 373    # ══════════════════════════════════════════════════════════════════════374    # Specialised handlers375    # ══════════════════════════════════════════════════════════════════════376 377    def _handle_conversational(self, question: str) -> QAResult:378        """Handle greetings, thanks, and casual messages without RAG."""379        response = _get_conversational_response(question)380        logger.info("Conversational message — responding without RAG.")381        return QAResult(382            answer=response,383            sources=[],384            chunks_used=0,385            model_used="none",386            intent=_QuestionIntent.CONVERSATIONAL,387        )388 389    def _stream_conversational(self, question: str):390        """Streaming variant for conversational messages."""391        response = _get_conversational_response(question)392        logger.info("[stream] Conversational message — instant reply.")393        yield {"token": response}394        yield {"done": True, "sources": [], "chunks_used": 0,395               "model": "none", "intent": _QuestionIntent.CONVERSATIONAL}396 397    def _handle_structure(self, question: str) -> QAResult:398        """Full file list injection — bypasses vector search entirely."""399        sources = self.all_sources400        context = (401            "The following is the complete list of all files indexed "402            "from this repository (one per line):\n\n"403            + "\n".join(sources)404            + "\n\n[INSTRUCTION: Present the file structure as a clean, "405            "indented directory tree. Group files by their directories. "406            "Add a brief one-line description for key files if their purpose "407            "is obvious from the name.]\n"408        )409        logger.info("Structure question — injecting %d file paths.", len(sources))410        answer_text = self.llm_client.ask(question=question, context=context)411        return QAResult(412            answer=answer_text,413            sources=sources[:20],414            chunks_used=len(sources),415            model_used=self.llm_client.model,416            intent=_QuestionIntent.STRUCTURE,417        )418 419    def _handle_architecture(self, question: str) -> QAResult:420        """Combine file list overview + broader chunk retrieval for architecture."""421        sources = self.all_sources422        file_overview = (423            "Complete file list for architectural context:\n"424            + "\n".join(sources) + "\n\n"425        )426 427        # Also retrieve semantic chunks for deeper context428        top_k = _INTENT_TOP_K[_QuestionIntent.ARCHITECTURE]429        query_vec = self.embedder.embed_query(question)430        results = self.vector_store.search(query_vec, top_k=top_k)431 432        chunk_context, chunk_sources, chunks_used = _build_context(433            results, MAX_CONTEXT_CHARS - len(file_overview)434        )435 436        context = (437            file_overview + chunk_context438            + _SCENARIO_HINTS[_QuestionIntent.ARCHITECTURE]439        )440 441        all_sources = list(dict.fromkeys(sources[:10] + chunk_sources))442        answer_text = self.llm_client.ask(question=question, context=context)443 444        return QAResult(445            answer=answer_text,446            sources=all_sources,447            chunks_used=chunks_used + len(sources),448            top_results=results,449            model_used=self.llm_client.model,450            intent=_QuestionIntent.ARCHITECTURE,451        )452 453    def _stream_structure(self, question: str):454        """Streaming variant of _handle_structure."""455        sources = self.all_sources456        context = (457            "The following is the complete list of all files indexed "458            "from this repository (one per line):\n\n"459            + "\n".join(sources)460            + "\n\n[INSTRUCTION: Present the file structure as a clean, "461            "indented directory tree. Group files by directories.]\n"462        )463        logger.info("[stream] Structure question — %d file paths.", len(sources))464        for token in self.llm_client.stream_tokens(question=question, context=context):465            yield {"token": token}466        yield {"done": True, "sources": sources[:20], "chunks_used": len(sources),467               "model": self.llm_client.model, "intent": _QuestionIntent.STRUCTURE}468 469    def _stream_architecture(self, question: str):470        """Streaming variant of _handle_architecture."""471        sources = self.all_sources472        file_overview = (473            "Complete file list for architectural context:\n"474            + "\n".join(sources) + "\n\n"475        )476        top_k = _INTENT_TOP_K[_QuestionIntent.ARCHITECTURE]477        query_vec = self.embedder.embed_query(question)478        results = self.vector_store.search(query_vec, top_k=top_k)479 480        chunk_context, chunk_sources, chunks_used = _build_context(481            results, MAX_CONTEXT_CHARS - len(file_overview)482        )483        context = (484            file_overview + chunk_context485            + _SCENARIO_HINTS[_QuestionIntent.ARCHITECTURE]486        )487        all_sources = list(dict.fromkeys(sources[:10] + chunk_sources))488 489        logger.info("[stream] Architecture question — %d paths + %d chunks.", len(sources), chunks_used)490        for token in self.llm_client.stream_tokens(question=question, context=context):491            yield {"token": token}492        yield {"done": True, "sources": all_sources,493               "chunks_used": chunks_used + len(sources),494               "model": self.llm_client.model, "intent": _QuestionIntent.ARCHITECTURE}495 496 497# ══════════════════════════════════════════════════════════════════════════════498# Conversational Response Helper499# ══════════════════════════════════════════════════════════════════════════════500 501def _get_conversational_response(question: str) -> str:502    """Return a friendly response for casual/conversational messages."""503    q = question.lower().strip().rstrip("?!.")504 505    # Thank you variants506    if any(w in q for w in ["thank", "thanks", "thankyou", "thx"]):507        return (508            "You're welcome! 😊 Feel free to ask anything else "509            "about the codebase — I'm here to help!"510        )511 512    # Greetings513    if any(w in q for w in ["hello", "hi", "hey", "good morning", "good evening", "good afternoon"]):514        return (515            "Hey there! 👋 I'm ready to help you explore this repository. "516            "Ask me anything about the code — how it works, what a file does, "517            "the project structure, or any specific question!"518        )519 520    # Farewells521    if any(w in q for w in ["bye", "goodbye", "see you", "take care"]):522        return "Goodbye! 👋 Happy coding, and feel free to come back anytime!"523 524    # Acknowledgements525    if any(w in q for w in ["ok", "okay", "got it", "understood", "makes sense"]):526        return (527            "Great! Let me know if you have any more questions "528            "about the codebase. 🚀"529        )530 531    # Positive feedback532    if any(w in q for w in ["great", "awesome", "perfect", "nice", "cool", "wonderful"]):533        return "Glad I could help! 😄 Ask away if you need anything else!"534 535    # Identity questions536    if any(w in q for w in ["who are you", "what are you", "what can you do"]):537        return (538            "I'm a **Code Explainer AI** 🤖 — I analyze GitHub repositories and answer "539            "your questions about the code. You can ask me about:\n\n"540            "- 📂 **File structure** — \"Show the project layout\"\n"541            "- 🏗️ **Architecture** — \"How does this project work?\"\n"542            "- 🔍 **Specific code** — \"What does auth.py do?\"\n"543            "- 🐛 **Debugging** — \"Why might this break?\"\n"544            "- 📦 **Dependencies** — \"What libraries does this use?\"\n\n"545            "Just ask in plain English!"546        )547 548    # Generic fallback for other conversational messages549    return (550        "I'm here to help you understand the code! "551        "Ask me any question about the repository. 😊"552    )553 554 555# ══════════════════════════════════════════════════════════════════════════════556# Context Builder557# ══════════════════════════════════════════════════════════════════════════════558 559def _build_context(560    results: list[dict],561    max_chars: int,562) -> tuple[str, list[str], int]:563    """564    Concatenate retrieved chunks into a single context string.565    Truncates to max_chars to stay inside the LLM context window.566    """567    parts:   list[str] = []568    sources: list[str] = []569    total   = 0570    used    = 0571 572    for r in results:573        source  = r["source"]574        score   = r["score"]575        text    = r["document"].page_content576        header  = f"# File: {source}  (relevance: {score:.2f})\n"577        block   = header + text + "\n\n---\n\n"578 579        if total + len(block) > max_chars:580            remaining = max_chars - total - len(header) - 10581            if remaining > 100:582                parts.append(header + text[:remaining] + "\n… [truncated]\n\n---\n\n")583                used += 1584            break585 586        parts.append(block)587        total += len(block)588        used  += 1589        if source not in sources:590            sources.append(source)591 592    return "".join(parts), sources, used593