Team Ai
Apppublic

Shashiguduri/github-code-explainer

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
helpers.py102 linesDownload Raw Back to utils
1"""2backend/utils/helpers.py3------------------------4URL parsing, file filtering, and path blocking utilities.5"""6 7import re8from urllib.parse import urlparse9 10# ---------------------------------------------------------------------------11# File extension allow-list12# ---------------------------------------------------------------------------13ALLOWED_EXTENSIONS = {14    ".py", ".js", ".ts", ".jsx", ".tsx",15    ".java", ".cpp", ".c", ".cs", ".go",16    ".rb", ".rs", ".php", ".swift", ".kt",17    ".scala", ".sh", ".bash",18}19 20# ---------------------------------------------------------------------------21# Directory block-list22# ---------------------------------------------------------------------------23IGNORED_DIRS = {24    # Build / package artefacts25    "node_modules", "dist", "build", ".git", "__pycache__",26    "venv", ".venv", "env", "vendor", "target", ".idea",27    ".vscode", "coverage", ".nyc_output", "out", "bin", "obj",28    "logs", "tmp", "temp", "cache", ".cache",29    # Tests — often 40%+ of chunks, rarely useful for code explanations30    "test", "tests", "__tests__", "spec", "specs", "__spec__",31    "e2e", "integration", "unit",32    # Docs / examples — low retrieval value33    "docs", "doc", "documentation",34    "examples", "example", "demo", "demos", "sample", "samples",35    "fixtures", "mocks", "stubs", "__mocks__",36    "migrations", "seeds",37}38 39# 200 KB max per file40MAX_FILE_SIZE_BYTES = 200_00041 42 43def parse_github_url(url: str) -> tuple[str, str]:44    """45    Extract (owner, repo) from a GitHub URL.46 47    Supports:48        https://github.com/owner/repo49        https://github.com/owner/repo.git50        https://github.com/owner/repo/tree/branch/...51        owner/repo  (shorthand)52 53    Raises:54        ValueError: if the URL cannot be parsed as a GitHub repo.55    """56    url = url.strip().rstrip("/")57 58    # Shorthand: owner/repo59    if re.fullmatch(r"[A-Za-z0-9_.-]+/[A-Za-z0-9_.-]+", url):60        owner, repo = url.split("/")61        return owner, repo.removesuffix(".git")62 63    parsed = urlparse(url)64    if "github.com" not in parsed.netloc:65        raise ValueError(66            f"Not a GitHub URL: '{url}'. "67            "Expected format: https://github.com/owner/repo"68        )69 70    parts = [p for p in parsed.path.split("/") if p]71    if len(parts) < 2:72        raise ValueError(73            f"Cannot extract owner/repo from: '{url}'. "74            "Expected format: https://github.com/owner/repo"75        )76 77    return parts[0], parts[1].removesuffix(".git")78 79 80def is_code_file(path: str) -> bool:81    """Return True if the file extension is in the allowed list."""82    dot = path.rfind(".")83    ext = path[dot:].lower() if dot != -1 else ""84    return ext in ALLOWED_EXTENSIONS85 86 87def is_ignored_path(path: str) -> bool:88    """Return True if any segment of the path matches the block-list."""89    parts = path.replace("\\", "/").split("/")90    return any(part in IGNORED_DIRS for part in parts)91 92 93def should_index_file(path: str, size: int | None = None) -> bool:94    """Combined check: extension + path + size."""95    if is_ignored_path(path):96        return False97    if not is_code_file(path):98        return False99    if size is not None and size > MAX_FILE_SIZE_BYTES:100        return False101    return True102