Team Ai
Apppublic

CamQuest/codesearch

sourceHugging Facemitupdated 3mo agoView on Hugging Face
0likes
data.py176 linesDownload Raw Back to codesearch
1"""2CodeSearchNet Python split loader.3 4Schema after loading:5    {6        "corpus":  List[dict]  — each dict has keys: id, code, docstring, url7        "queries": List[dict]  — each dict has keys: id, query, relevant_id8    }9 10Document ID is `func_code_url` — globally unique (encodes repo + commit + file + lines).11Using `func_name` as ID is wrong: the same name (e.g. `parse`, `get`) appears in hundreds12of unrelated repos, and train/test splits come from completely different codebases.13 14Smoke-test mode (0 < n):15    Corpus  = first n docs from the test split.16    Queries = the non-empty docstrings from those same n docs.17    Every query has a valid relevant doc in the corpus. BM25 will score near-perfectly18    here (query == indexed docstring) — this mode only validates the pipeline, not quality.19 20Full-corpus mode (n == -1):21    Corpus  = train split (~412k) + test split (~22k) = ~434k total.22    Queries = test split docstrings (~22k, filtered for non-empty).23    Each test function IS in the corpus (URL uniqueness confirmed: zero overlap24    between train and test URLs), so every query has valid ground truth.25    BM25 must find the relevant doc among ~434k candidates.26"""27 28from __future__ import annotations29 30import ast31import textwrap32 33from datasets import load_dataset34from tqdm import tqdm35 36from codesearch.config import SMOKE_TEST_SIZE37 38 39def _make_doc(row: dict) -> dict:40    url = row["func_code_url"]41    return {42        "id": url,43        "code": row["whole_func_string"],44        "code_tokens": " ".join(row["func_code_tokens"]),45        "docstring": row["func_documentation_string"],46        "url": url,47    }48 49 50def strip_docstring(src: str) -> str | None:51    """Return the function source with its docstring removed, or None if the52    source can't be parsed (Python-2 / malformed rows — caller should fall back).53 54    AST span-excision: locate the leading string-literal statement of the function55    body and cut exactly its character span out of the (dedented) source. This56    preserves comments, formatting, and string literals — unlike `code_tokens`,57    which drops comments and mangles spacing.58 59    This is the dense-embedding input for the M5 "embed source vs code_tokens"60    experiment. The docstring MUST be removed: the eval query *is* the docstring,61    so embedding it raw would be a trivial self-match (leakage → fake MRR ~0.9).62    Not wired into _make_doc — it's called only by the offline indexer, so we63    don't pay AST parsing on every corpus load.64    """65    try:66        ded = textwrap.dedent(src)67        tree = ast.parse(ded)68    except SyntaxError:69        return None70    node = tree.body[0] if tree.body else None71    if not isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):72        return ded73    b0 = node.body[0] if node.body else None74    if (75        isinstance(b0, ast.Expr)76        and isinstance(b0.value, ast.Constant)77        and isinstance(b0.value.value, str)78    ):79        lines = ded.splitlines(keepends=True)80        sl, sc = b0.lineno - 1, b0.col_offset81        el, ec = b0.end_lineno - 1, b0.end_col_offset82        if sl == el:83            lines[sl] = lines[sl][:sc] + lines[sl][ec:]84        else:85            lines[sl] = lines[sl][:sc]86            for k in range(sl + 1, el):87                lines[k] = ""88            lines[el] = lines[el][ec:]89        return "".join(lines)90    return ded91 92 93def load_codesearch(94    n: int = SMOKE_TEST_SIZE,95    queries_only: bool = False,96) -> tuple[list[dict], list[dict]]:97    """98    Load CodeSearchNet Python split.99 100    Args:101        n: -1 = full corpus (train + test, ~434k docs).102           n > 0 = smoke-test: first n docs from test split only.103        queries_only: if True, skip building the corpus list (saves the ~412k train104                      split scan). Returns ([], queries). Use when only the eval105                      queries are needed and the corpus already lives in an external106                      store (e.g., Qdrant). Only valid with n=-1.107 108    Returns:109        (corpus, queries) — see module docstring for schema.110        When queries_only=True, corpus is [].111    """112    raw_test = load_dataset("code-search-net/code_search_net", "python", split="test")113 114    if queries_only:115        if n != -1:116            raise ValueError("queries_only=True requires n=-1 (full corpus mode).")117        print("Loading CodeSearchNet Python split (queries only — skipping corpus)...")118        queries: list[dict] = []119        for row in tqdm(raw_test, desc="Building eval queries from test docstrings"):120            url = row["func_code_url"]121            if not url:122                continue123            text = row["func_documentation_string"]124            if not text or not text.strip():125                continue126            queries.append({"id": f"q_{url}", "query": text, "relevant_id": url})127        print(f"Loaded {len(queries):,} eval queries (corpus skipped).")128        return [], queries129 130    corpus: list[dict] = []131    seen_urls: set[str] = set()132 133    if n != -1:134        print(f"Loading CodeSearchNet Python split (smoke-test, n={n})...")135        for row in tqdm(raw_test, desc="Adding test split to corpus (smoke-test)"):136            if len(corpus) >= n:137                break138            url = row["func_code_url"]139            if not url or url in seen_urls:140                continue141            seen_urls.add(url)142            corpus.append(_make_doc(row))143    else:144        print("Loading CodeSearchNet Python split (full corpus)...")145        raw_train = load_dataset("code-search-net/code_search_net", "python", split="train")146        for split_name, raw in [("train", raw_train), ("test", raw_test)]:147            for row in tqdm(raw, desc=f"Adding {split_name} split to corpus"):148                url = row["func_code_url"]149                if not url or url in seen_urls:150                    continue151                seen_urls.add(url)152                corpus.append(_make_doc(row))153 154    corpus_ids = {doc["id"] for doc in corpus}155 156    # Queries: test-split docstrings whose function is in the corpus.157    # relevant_id = func_code_url, which is the doc["id"] of the target function.158    queries: list[dict] = []159    for row in tqdm(raw_test, desc="Building eval queries from test docstrings"):160        url = row["func_code_url"]161        if url not in corpus_ids:162            continue163        query_text = row["func_documentation_string"]164        if not query_text or not query_text.strip():165            continue166        queries.append(167            {168                "id": f"q_{url}",169                "query": query_text,170                "relevant_id": url,171            }172        )173 174    print(f"Loaded {len(corpus):,} corpus docs, {len(queries):,} eval queries.")175    return corpus, queries176