Team Ai
Apppublic

Shashiguduri/github-code-explainer

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
chunker.py116 linesDownload Raw Back to services
1"""2backend/services/chunker.py3-----------------------------4Splits raw code files into overlapping chunks using LangChain's5language-aware RecursiveCharacterTextSplitter.6 7Language-aware splitting prefers logical boundaries (functions, classes)8before falling back to character-level splits — producing higher-quality9chunks for code retrieval.10"""11 12import logging13from langchain_text_splitters import RecursiveCharacterTextSplitter, Language14from langchain_core.documents import Document15 16logger = logging.getLogger(__name__)17 18# Extension → LangChain Language enum19EXT_TO_LANG: dict[str, Language] = {20    ".py":   Language.PYTHON,21    ".js":   Language.JS,22    ".jsx":  Language.JS,23    ".ts":   Language.JS,24    ".tsx":  Language.JS,25    ".java": Language.JAVA,26    ".cpp":  Language.CPP,27    ".c":    Language.CPP,28    ".go":   Language.GO,29    ".rb":   Language.RUBY,30    ".rs":   Language.RUST,31    ".scala":Language.SCALA,32}33 34 35def _ext(path: str) -> str:36    dot = path.rfind(".")37    return path[dot:].lower() if dot != -1 else ""38 39 40def _make_splitter(41    path: str, chunk_size: int, chunk_overlap: int42) -> RecursiveCharacterTextSplitter:43    lang = EXT_TO_LANG.get(_ext(path))44    if lang:45        return RecursiveCharacterTextSplitter.from_language(46            language=lang,47            chunk_size=chunk_size,48            chunk_overlap=chunk_overlap,49        )50    return RecursiveCharacterTextSplitter(51        chunk_size=chunk_size,52        chunk_overlap=chunk_overlap,53        length_function=len,54    )55 56 57def chunk_files(58    files: list[dict],59    chunk_size: int = 1500,60    chunk_overlap: int = 75,61) -> list[Document]:62    """63    Convert raw file dicts into LangChain Documents.64 65    Args:66        files:        Output of repo_loader.load_repository().67                      Each: {"path": str, "content": str}68        chunk_size:   Max characters per chunk.69        chunk_overlap: Shared characters at chunk boundaries.70 71    Returns:72        list[Document] — each with metadata:73            source, extension, chunk_index, total_chunks74    """75    all_docs: list[Document] = []76    skipped = 077 78    for file in files:79        path: str    = file["path"]80        content: str = file["content"]81 82        if not content or not content.strip():83            skipped += 184            continue85 86        splitter = _make_splitter(path, chunk_size, chunk_overlap)87 88        try:89            chunks = splitter.split_text(content)90        except Exception as exc:91            logger.warning("Splitter error on %s: %s — skipping.", path, exc)92            skipped += 193            continue94 95        # Drop trivially small chunks96        chunks = [c for c in chunks if len(c.strip()) > 20]97        total  = len(chunks)98        ext    = _ext(path)99 100        for i, text in enumerate(chunks):101            all_docs.append(Document(102                page_content=text,103                metadata={104                    "source":       path,105                    "extension":    ext,106                    "chunk_index":  i,107                    "total_chunks": total,108                },109            ))110 111    logger.info(112        "Chunking done: %d files → %d chunks (%d skipped).",113        len(files) - skipped, len(all_docs), skipped,114    )115    return all_docs116