Shashiguduri/github-code-explainer
0
1"""2backend/services/repo_loader.py3--------------------------------4Fetches source code files from a public GitHub repository.5 6Optimised for free-tier cloud hosting (Render, Railway, Fly.io):7 - Uses raw.githubusercontent.com instead of /contents API8 → no base64 decode, no JSON parsing, smaller payloads, faster9 - Concurrent downloads via ThreadPoolExecutor (10 workers)10 - Token auth supported on raw URLs too (same header)11 12Flow:13 1. Parse owner + repo from URL (0 API calls)14 2. Detect default branch (1 API call)15 3. Fetch full recursive file tree (1 API call)16 4. Filter files via should_index_file()17 5. Download raw content concurrently18 6. Return list[dict] with path + content19"""20 21import os22import logging23import requests24from concurrent.futures import ThreadPoolExecutor, as_completed25from utils.helpers import parse_github_url, should_index_file26 27logger = logging.getLogger(__name__)28 29API_BASE = "https://api.github.com"30RAW_BASE = "https://raw.githubusercontent.com"31MAX_WORKERS = 2032 33 34def _auth_headers() -> dict:35 """Return auth headers if GITHUB_TOKEN is set, otherwise empty."""36 token = os.getenv("GITHUB_TOKEN", "").strip()37 if token:38 return {"Authorization": f"Bearer {token}"}39 return {}40 41 42def _build_api_session() -> requests.Session:43 """Session for GitHub REST API calls (tree, branch)."""44 s = requests.Session()45 token = os.getenv("GITHUB_TOKEN", "").strip()46 if token:47 s.headers.update({"Authorization": f"Bearer {token}"})48 logger.info("GitHub: using token auth (5000 req/hr limit).")49 else:50 logger.warning("GITHUB_TOKEN not set — 60 req/hr limit applies.")51 s.headers.update({"Accept": "application/vnd.github+json"})52 return s53 54 55def _get_default_branch(session: requests.Session, owner: str, repo: str) -> str:56 resp = session.get(f"{API_BASE}/repos/{owner}/{repo}")57 if resp.status_code == 404:58 raise ValueError(f"Repository not found or is private: {owner}/{repo}")59 resp.raise_for_status()60 return resp.json().get("default_branch", "main")61 62 63def _get_file_tree(64 session: requests.Session, owner: str, repo: str, branch: str65) -> list[dict]:66 url = f"{API_BASE}/repos/{owner}/{repo}/git/trees/{branch}?recursive=1"67 resp = session.get(url)68 resp.raise_for_status()69 data = resp.json()70 if data.get("truncated"):71 logger.warning("Tree is truncated — repo is very large.")72 return data.get("tree", [])73 74 75def _fetch_raw(owner: str, repo: str, branch: str, path: str) -> str | None:76 """77 Fetch file content directly from raw.githubusercontent.com.78 79 Faster than the /contents API because:80 - Returns plain text (no JSON wrapper)81 - No base64 encoding/decoding82 - Smaller HTTP response payload83 """84 url = f"{RAW_BASE}/{owner}/{repo}/{branch}/{path}"85 try:86 resp = requests.get(url, headers=_auth_headers(), timeout=15)87 if resp.status_code == 404:88 return None89 resp.raise_for_status()90 return resp.text91 except Exception as exc:92 logger.warning("Failed fetching %s: %s", path, exc)93 return None94 95 96def load_repository(97 repo_url: str,98 progress_callback=None,99 max_workers: int = MAX_WORKERS,100) -> list[dict]:101 """102 Load all indexable code files from a GitHub repository.103 104 Cloud-optimised: uses raw content URLs + concurrent downloads.105 106 Args:107 repo_url: Full GitHub URL or 'owner/repo' shorthand.108 max_workers: Concurrent download threads (default 10).109 110 Returns:111 list[{"path": str, "content": str}]112 """113 owner, repo = parse_github_url(repo_url)114 logger.info("Loading repo: %s/%s", owner, repo)115 116 api_session = _build_api_session()117 branch = _get_default_branch(api_session, owner, repo)118 tree = _get_file_tree(api_session, owner, repo, branch)119 120 code_nodes = [121 n for n in tree122 if n["type"] == "blob" and should_index_file(n["path"], n.get("size"))123 ]124 125 if not code_nodes:126 raise ValueError(127 f"No indexable code files found in '{owner}/{repo}'."128 )129 130 total = len(code_nodes)131 logger.info(132 "Fetching %d files from raw.githubusercontent.com (%d workers) ...",133 total, max_workers,134 )135 136 # ── Concurrent raw content download ────────────────────────────────────137 def fetch_one(node: dict) -> tuple[str, str | None]:138 path = node["path"]139 content = _fetch_raw(owner, repo, branch, path)140 return path, content141 142 completed = 0143 path_to_content: dict[str, str] = {}144 145 with ThreadPoolExecutor(max_workers=max_workers) as executor:146 futures = {executor.submit(fetch_one, n): n for n in code_nodes}147 for future in as_completed(futures):148 path, content = future.result()149 completed += 1150 if progress_callback:151 progress_callback(completed, total, path)152 if content and content.strip():153 path_to_content[path] = content154 155 # Restore original tree order156 files = [157 {"path": n["path"], "content": path_to_content[n["path"]]}158 for n in code_nodes159 if n["path"] in path_to_content160 ]161 162 logger.info("Download complete: %d/%d files.", len(files), total)163 return files164 