Team Ai
Apppublic

Shashiguduri/github-code-explainer

sourceHugging Faceupdated 7mo agoView on Hugging Face
0likes
repo_loader.py164 linesDownload Raw Back to services
1"""2backend/services/repo_loader.py3--------------------------------4Fetches source code files from a public GitHub repository.5 6Optimised for free-tier cloud hosting (Render, Railway, Fly.io):7  - Uses raw.githubusercontent.com instead of /contents API8    → no base64 decode, no JSON parsing, smaller payloads, faster9  - Concurrent downloads via ThreadPoolExecutor (10 workers)10  - Token auth supported on raw URLs too (same header)11 12Flow:13  1. Parse owner + repo from URL            (0 API calls)14  2. Detect default branch                  (1 API call)15  3. Fetch full recursive file tree         (1 API call)16  4. Filter files via should_index_file()17  5. Download raw content concurrently18  6. Return list[dict] with path + content19"""20 21import os22import logging23import requests24from concurrent.futures import ThreadPoolExecutor, as_completed25from utils.helpers import parse_github_url, should_index_file26 27logger = logging.getLogger(__name__)28 29API_BASE     = "https://api.github.com"30RAW_BASE     = "https://raw.githubusercontent.com"31MAX_WORKERS  = 2032 33 34def _auth_headers() -> dict:35    """Return auth headers if GITHUB_TOKEN is set, otherwise empty."""36    token = os.getenv("GITHUB_TOKEN", "").strip()37    if token:38        return {"Authorization": f"Bearer {token}"}39    return {}40 41 42def _build_api_session() -> requests.Session:43    """Session for GitHub REST API calls (tree, branch)."""44    s = requests.Session()45    token = os.getenv("GITHUB_TOKEN", "").strip()46    if token:47        s.headers.update({"Authorization": f"Bearer {token}"})48        logger.info("GitHub: using token auth (5000 req/hr limit).")49    else:50        logger.warning("GITHUB_TOKEN not set — 60 req/hr limit applies.")51    s.headers.update({"Accept": "application/vnd.github+json"})52    return s53 54 55def _get_default_branch(session: requests.Session, owner: str, repo: str) -> str:56    resp = session.get(f"{API_BASE}/repos/{owner}/{repo}")57    if resp.status_code == 404:58        raise ValueError(f"Repository not found or is private: {owner}/{repo}")59    resp.raise_for_status()60    return resp.json().get("default_branch", "main")61 62 63def _get_file_tree(64    session: requests.Session, owner: str, repo: str, branch: str65) -> list[dict]:66    url  = f"{API_BASE}/repos/{owner}/{repo}/git/trees/{branch}?recursive=1"67    resp = session.get(url)68    resp.raise_for_status()69    data = resp.json()70    if data.get("truncated"):71        logger.warning("Tree is truncated — repo is very large.")72    return data.get("tree", [])73 74 75def _fetch_raw(owner: str, repo: str, branch: str, path: str) -> str | None:76    """77    Fetch file content directly from raw.githubusercontent.com.78 79    Faster than the /contents API because:80      - Returns plain text (no JSON wrapper)81      - No base64 encoding/decoding82      - Smaller HTTP response payload83    """84    url = f"{RAW_BASE}/{owner}/{repo}/{branch}/{path}"85    try:86        resp = requests.get(url, headers=_auth_headers(), timeout=15)87        if resp.status_code == 404:88            return None89        resp.raise_for_status()90        return resp.text91    except Exception as exc:92        logger.warning("Failed fetching %s: %s", path, exc)93        return None94 95 96def load_repository(97    repo_url: str,98    progress_callback=None,99    max_workers: int = MAX_WORKERS,100) -> list[dict]:101    """102    Load all indexable code files from a GitHub repository.103 104    Cloud-optimised: uses raw content URLs + concurrent downloads.105 106    Args:107        repo_url:     Full GitHub URL or 'owner/repo' shorthand.108        max_workers:  Concurrent download threads (default 10).109 110    Returns:111        list[{"path": str, "content": str}]112    """113    owner, repo = parse_github_url(repo_url)114    logger.info("Loading repo: %s/%s", owner, repo)115 116    api_session = _build_api_session()117    branch      = _get_default_branch(api_session, owner, repo)118    tree        = _get_file_tree(api_session, owner, repo, branch)119 120    code_nodes = [121        n for n in tree122        if n["type"] == "blob" and should_index_file(n["path"], n.get("size"))123    ]124 125    if not code_nodes:126        raise ValueError(127            f"No indexable code files found in '{owner}/{repo}'."128        )129 130    total = len(code_nodes)131    logger.info(132        "Fetching %d files from raw.githubusercontent.com (%d workers) ...",133        total, max_workers,134    )135 136    # ── Concurrent raw content download ────────────────────────────────────137    def fetch_one(node: dict) -> tuple[str, str | None]:138        path    = node["path"]139        content = _fetch_raw(owner, repo, branch, path)140        return path, content141 142    completed        = 0143    path_to_content: dict[str, str] = {}144 145    with ThreadPoolExecutor(max_workers=max_workers) as executor:146        futures = {executor.submit(fetch_one, n): n for n in code_nodes}147        for future in as_completed(futures):148            path, content = future.result()149            completed += 1150            if progress_callback:151                progress_callback(completed, total, path)152            if content and content.strip():153                path_to_content[path] = content154 155    # Restore original tree order156    files = [157        {"path": n["path"], "content": path_to_content[n["path"]]}158        for n in code_nodes159        if n["path"] in path_to_content160    ]161 162    logger.info("Download complete: %d/%d files.", len(files), total)163    return files164