CamQuest/codesearch
0
1"""2CodeSearchNet Python split loader.3 4Schema after loading:5 {6 "corpus": List[dict] — each dict has keys: id, code, docstring, url7 "queries": List[dict] — each dict has keys: id, query, relevant_id8 }9 10Document ID is `func_code_url` — globally unique (encodes repo + commit + file + lines).11Using `func_name` as ID is wrong: the same name (e.g. `parse`, `get`) appears in hundreds12of unrelated repos, and train/test splits come from completely different codebases.13 14Smoke-test mode (0 < n):15 Corpus = first n docs from the test split.16 Queries = the non-empty docstrings from those same n docs.17 Every query has a valid relevant doc in the corpus. BM25 will score near-perfectly18 here (query == indexed docstring) — this mode only validates the pipeline, not quality.19 20Full-corpus mode (n == -1):21 Corpus = train split (~412k) + test split (~22k) = ~434k total.22 Queries = test split docstrings (~22k, filtered for non-empty).23 Each test function IS in the corpus (URL uniqueness confirmed: zero overlap24 between train and test URLs), so every query has valid ground truth.25 BM25 must find the relevant doc among ~434k candidates.26"""27 28from __future__ import annotations29 30import ast31import textwrap32 33from datasets import load_dataset34from tqdm import tqdm35 36from codesearch.config import SMOKE_TEST_SIZE37 38 39def _make_doc(row: dict) -> dict:40 url = row["func_code_url"]41 return {42 "id": url,43 "code": row["whole_func_string"],44 "code_tokens": " ".join(row["func_code_tokens"]),45 "docstring": row["func_documentation_string"],46 "url": url,47 }48 49 50def strip_docstring(src: str) -> str | None:51 """Return the function source with its docstring removed, or None if the52 source can't be parsed (Python-2 / malformed rows — caller should fall back).53 54 AST span-excision: locate the leading string-literal statement of the function55 body and cut exactly its character span out of the (dedented) source. This56 preserves comments, formatting, and string literals — unlike `code_tokens`,57 which drops comments and mangles spacing.58 59 This is the dense-embedding input for the M5 "embed source vs code_tokens"60 experiment. The docstring MUST be removed: the eval query *is* the docstring,61 so embedding it raw would be a trivial self-match (leakage → fake MRR ~0.9).62 Not wired into _make_doc — it's called only by the offline indexer, so we63 don't pay AST parsing on every corpus load.64 """65 try:66 ded = textwrap.dedent(src)67 tree = ast.parse(ded)68 except SyntaxError:69 return None70 node = tree.body[0] if tree.body else None71 if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):72 return ded73 b0 = node.body[0] if node.body else None74 if (75 isinstance(b0, ast.Expr)76 and isinstance(b0.value, ast.Constant)77 and isinstance(b0.value.value, str)78 ):79 lines = ded.splitlines(keepends=True)80 sl, sc = b0.lineno - 1, b0.col_offset81 el, ec = b0.end_lineno - 1, b0.end_col_offset82 if sl == el:83 lines[sl] = lines[sl][:sc] + lines[sl][ec:]84 else:85 lines[sl] = lines[sl][:sc]86 for k in range(sl + 1, el):87 lines[k] = ""88 lines[el] = lines[el][ec:]89 return "".join(lines)90 return ded91 92 93def load_codesearch(94 n: int = SMOKE_TEST_SIZE,95 queries_only: bool = False,96) -> tuple[list[dict], list[dict]]:97 """98 Load CodeSearchNet Python split.99 100 Args:101 n: -1 = full corpus (train + test, ~434k docs).102 n > 0 = smoke-test: first n docs from test split only.103 queries_only: if True, skip building the corpus list (saves the ~412k train104 split scan). Returns ([], queries). Use when only the eval105 queries are needed and the corpus already lives in an external106 store (e.g., Qdrant). Only valid with n=-1.107 108 Returns:109 (corpus, queries) — see module docstring for schema.110 When queries_only=True, corpus is [].111 """112 raw_test = load_dataset("code-search-net/code_search_net", "python", split="test")113 114 if queries_only:115 if n != -1:116 raise ValueError("queries_only=True requires n=-1 (full corpus mode).")117 print("Loading CodeSearchNet Python split (queries only — skipping corpus)...")118 queries: list[dict] = []119 for row in tqdm(raw_test, desc="Building eval queries from test docstrings"):120 url = row["func_code_url"]121 if not url:122 continue123 text = row["func_documentation_string"]124 if not text or not text.strip():125 continue126 queries.append({"id": f"q_{url}", "query": text, "relevant_id": url})127 print(f"Loaded {len(queries):,} eval queries (corpus skipped).")128 return [], queries129 130 corpus: list[dict] = []131 seen_urls: set[str] = set()132 133 if n != -1:134 print(f"Loading CodeSearchNet Python split (smoke-test, n={n})...")135 for row in tqdm(raw_test, desc="Adding test split to corpus (smoke-test)"):136 if len(corpus) >= n:137 break138 url = row["func_code_url"]139 if not url or url in seen_urls:140 continue141 seen_urls.add(url)142 corpus.append(_make_doc(row))143 else:144 print("Loading CodeSearchNet Python split (full corpus)...")145 raw_train = load_dataset("code-search-net/code_search_net", "python", split="train")146 for split_name, raw in [("train", raw_train), ("test", raw_test)]:147 for row in tqdm(raw, desc=f"Adding {split_name} split to corpus"):148 url = row["func_code_url"]149 if not url or url in seen_urls:150 continue151 seen_urls.add(url)152 corpus.append(_make_doc(row))153 154 corpus_ids = {doc["id"] for doc in corpus}155 156 # Queries: test-split docstrings whose function is in the corpus.157 # relevant_id = func_code_url, which is the doc["id"] of the target function.158 queries: list[dict] = []159 for row in tqdm(raw_test, desc="Building eval queries from test docstrings"):160 url = row["func_code_url"]161 if url not in corpus_ids:162 continue163 query_text = row["func_documentation_string"]164 if not query_text or not query_text.strip():165 continue166 queries.append(167 {168 "id": f"q_{url}",169 "query": query_text,170 "relevant_id": url,171 }172 )173 174 print(f"Loaded {len(corpus):,} corpus docs, {len(queries):,} eval queries.")175 return corpus, queries176 