Team Ai
Apppublic

mathidot111/First_agent_template

sourceHugging Faceupdated 5mo agoView on Hugging Face
0likes
rag_eval.py736 linesDownload Raw Back to eval
1from __future__ import annotations2 3import argparse4import csv5import json6import math7import re8import shutil9import zipfile10from dataclasses import dataclass11from pathlib import Path12from typing import Any, Iterable13 14import chromadb15import requests16from llama_index.core import StorageContext, VectorStoreIndex17from llama_index.core.node_parser import SentenceSplitter18from llama_index.core.schema import Document19from llama_index.core.schema import NodeWithScore, TextNode20from llama_index.vector_stores.chroma import ChromaVectorStore21 22from tools.query_knowledge import (23    BM25Retriever,24    EMBED_MODEL_NAME,25    RERANKER_MODEL_NAME,26    CrossEncoderReranker,27    configure_model_cache,28    resolve_embed_model_name,29)30 31 32PROJECT_ROOT = Path(__file__).resolve().parents[1]33EVAL_DIR = PROJECT_ROOT / "eval"34DATA_DIR = EVAL_DIR / "data"35INDEX_DIR = EVAL_DIR / "indexes"36REPORT_DIR = EVAL_DIR / "reports"37 38BEIR_URLS = {39    "scifact": "https://public.ukp.informatik.tu-darmstadt.de/thakur/BEIR/datasets/scifact.zip",40    "fiqa": "https://public.ukp.informatik.tu-darmstadt.de/thakur/BEIR/datasets/fiqa.zip",41}42 43DATASET_ALIASES = {44    "beir/scifact": "scifact",45    "beir/fiqa": "fiqa",46    "open-ragbench": "open_ragbench",47    "open_ragbench": "open_ragbench",48    "t2-ragbench": "t2_ragbench",49    "t2_ragbench": "t2_ragbench",50    "local-options": "local_options",51    "local_options": "local_options",52}53 54 55@dataclass56class EvalCorpus:57    name: str58    documents: list[dict[str, Any]]59    queries: list[dict[str, Any]]60    qrels: dict[str, set[str]]61 62 63def ensure_dirs() -> None:64    DATA_DIR.mkdir(parents=True, exist_ok=True)65    INDEX_DIR.mkdir(parents=True, exist_ok=True)66    REPORT_DIR.mkdir(parents=True, exist_ok=True)67 68 69def download_file(url: str, destination: Path) -> None:70    destination.parent.mkdir(parents=True, exist_ok=True)71    with requests.get(url, stream=True, timeout=60) as response:72        response.raise_for_status()73        with destination.open("wb") as file:74            for chunk in response.iter_content(chunk_size=1024 * 1024):75                if chunk:76                    file.write(chunk)77 78 79def read_jsonl(path: Path) -> Iterable[dict[str, Any]]:80    with path.open("r", encoding="utf-8") as file:81        for line in file:82            line = line.strip()83            if line:84                yield json.loads(line)85 86 87def prepare_beir_dataset(dataset_name: str) -> Path:88    ensure_dirs()89    if dataset_name not in BEIR_URLS:90        raise ValueError(f"Unsupported BEIR dataset: {dataset_name}")91 92    target_dir = DATA_DIR / "beir" / dataset_name93    corpus_path = target_dir / "corpus.jsonl"94    if corpus_path.exists():95        return target_dir96 97    zip_path = DATA_DIR / "downloads" / f"{dataset_name}.zip"98    if not zip_path.exists():99        download_file(BEIR_URLS[dataset_name], zip_path)100 101    extract_root = DATA_DIR / "beir"102    extract_root.mkdir(parents=True, exist_ok=True)103    with zipfile.ZipFile(zip_path) as archive:104        archive.extractall(extract_root)105 106    if not corpus_path.exists():107        raise FileNotFoundError(f"BEIR extraction did not create {corpus_path}")108 109    return target_dir110 111 112def load_beir_dataset(113    dataset_name: str,114    split: str,115    max_corpus_docs: int | None,116    max_queries: int | None,117) -> EvalCorpus:118    dataset_dir = prepare_beir_dataset(dataset_name)119 120    all_queries = {121        str(row["_id"]): row.get("text", "")122        for row in read_jsonl(dataset_dir / "queries.jsonl")123    }124 125    qrels_path = dataset_dir / "qrels" / f"{split}.tsv"126    if not qrels_path.exists():127        candidates = sorted((dataset_dir / "qrels").glob("*.tsv"))128        if not candidates:129            raise FileNotFoundError(f"No qrels found under {dataset_dir / 'qrels'}")130        qrels_path = candidates[0]131 132    all_qrels: dict[str, set[str]] = {}133    with qrels_path.open("r", encoding="utf-8") as file:134        reader = csv.DictReader(file, delimiter="\t")135        for row in reader:136            query_id = str(row.get("query-id") or row.get("query_id"))137            corpus_id = str(row.get("corpus-id") or row.get("corpus_id"))138            score = int(row.get("score", 1))139            if score <= 0:140                continue141            all_qrels.setdefault(query_id, set()).add(corpus_id)142 143    queries = []144    required_doc_ids = set()145    for query_id, relevant_docs in all_qrels.items():146        if query_id not in all_queries:147            continue148        if max_corpus_docs and len(required_doc_ids | relevant_docs) > max_corpus_docs:149            continue150        required_doc_ids.update(relevant_docs)151        queries.append(152            {153                "query_id": query_id,154                "question": all_queries[query_id],155                "relevant_doc_ids": sorted(relevant_docs),156            }157        )158        if max_queries and len(queries) >= max_queries:159            break160 161    documents = []162    seen_doc_ids = set()163    for row in read_jsonl(dataset_dir / "corpus.jsonl"):164        doc_id = str(row["_id"])165        if required_doc_ids and doc_id not in required_doc_ids:166            if max_corpus_docs and len(documents) >= max_corpus_docs:167                continue168            if max_corpus_docs and len(documents) + len(required_doc_ids - seen_doc_ids) >= max_corpus_docs:169                continue170        title = row.get("title") or ""171        text = row.get("text") or ""172        documents.append(173            {174                "doc_id": doc_id,175                "title": title,176                "text": f"{title}\n{text}".strip(),177                "metadata": {"source_dataset": f"beir/{dataset_name}"},178            }179        )180        seen_doc_ids.add(doc_id)181        if max_corpus_docs and len(documents) >= max_corpus_docs and required_doc_ids.issubset(seen_doc_ids):182            break183 184    if not documents or not queries:185        raise ValueError(186            f"Dataset beir/{dataset_name} has no evaluable documents/queries. "187            "Increase --max-corpus-docs or use a larger sample."188        )189 190    return EvalCorpus(191        name=f"beir_{dataset_name}",192        documents=documents,193        queries=queries,194        qrels={query["query_id"]: set(query["relevant_doc_ids"]) for query in queries},195    )196 197 198def snapshot_hf_dataset(repo_id: str, local_name: str) -> Path:199    from huggingface_hub import snapshot_download200 201    ensure_dirs()202    target_dir = DATA_DIR / "hf" / local_name203    if target_dir.exists():204        return target_dir205 206    snapshot_download(207        repo_id=repo_id,208        repo_type="dataset",209        local_dir=str(target_dir),210        local_dir_use_symlinks=False,211    )212    return target_dir213 214 215def flatten_open_ragbench_section(section: dict[str, Any]) -> str:216    parts = [section.get("text") or ""]217    tables = section.get("tables") or {}218    if isinstance(tables, dict):219        parts.extend(str(value) for value in tables.values())220    return "\n".join(part for part in parts if part)221 222 223def load_open_ragbench(224    max_corpus_docs: int | None,225    max_queries: int | None,226) -> EvalCorpus:227    dataset_dir = snapshot_hf_dataset("vectara/open_ragbench", "open_ragbench")228    root = dataset_dir / "pdf" / "arxiv"229    if not root.exists():230        root = dataset_dir / "official" / "pdf" / "arxiv"231    if not root.exists():232        raise FileNotFoundError(f"Open RAGBench root not found: {root}")233 234    queries_data = json.loads((root / "queries.json").read_text(encoding="utf-8"))235    qrels_data = json.loads((root / "qrels.json").read_text(encoding="utf-8"))236 237    documents = []238    qrels: dict[str, set[str]] = {}239    required_doc_ids = set()240    selected_query_ids = []241    for query_id, qrel in qrels_data.items():242        doc_id = str(qrel.get("doc_id"))243        if not doc_id or doc_id == "None":244            continue245        selected_query_ids.append(str(query_id))246        required_doc_ids.add(doc_id)247        if max_queries and len(selected_query_ids) >= max_queries:248            break249 250    allowed_doc_ids = set()251    corpus_files = sorted((root / "corpus").glob("*.json"))252 253    for corpus_file in corpus_files:254        paper = json.loads(corpus_file.read_text(encoding="utf-8"))255        paper_id = str(paper.get("id") or corpus_file.stem)256        is_required = paper_id in required_doc_ids257        if max_corpus_docs and not is_required:258            missing_required_count = len(required_doc_ids - allowed_doc_ids)259            if len(documents) + missing_required_count >= max_corpus_docs:260                continue261        allowed_doc_ids.add(paper_id)262        section_texts = []263        for section_index, section in enumerate(paper.get("sections") or []):264            section_text = flatten_open_ragbench_section(section)265            if section_text:266                section_texts.append(f"[section {section_index}]\n{section_text}")267        text = "\n\n".join(268            part269            for part in [paper.get("title") or "", paper.get("abstract") or "", *section_texts]270            if part271        )272        documents.append(273            {274                "doc_id": paper_id,275                "title": paper.get("title") or paper_id,276                "text": text,277                "metadata": {278                    "source_dataset": "open_ragbench",279                    "categories": ",".join(paper.get("categories") or []),280                },281            }282        )283        if max_corpus_docs and len(documents) >= max_corpus_docs:284            break285 286    queries = []287    for query_id in selected_query_ids:288        qrel = qrels_data[query_id]289        doc_id = str(qrel.get("doc_id"))290        if doc_id not in allowed_doc_ids:291            continue292        query_payload = queries_data.get(query_id) or {}293        question = query_payload.get("query") if isinstance(query_payload, dict) else str(query_payload)294        qrels[str(query_id)] = {doc_id}295        queries.append(296            {297                "query_id": str(query_id),298                "question": question,299                "relevant_doc_ids": [doc_id],300            }301        )302        if max_queries and len(queries) >= max_queries:303            break304 305    if not documents or not queries:306        raise ValueError("Open RAGBench produced no evaluable sample.")307 308    return EvalCorpus("open_ragbench", documents, queries, qrels)309 310 311def load_t2_ragbench(312    max_corpus_docs: int | None,313    max_queries: int | None,314) -> EvalCorpus:315    dataset_dir = snapshot_hf_dataset("G4KMU/t2-ragbench", "t2_ragbench")316    parquet_files = sorted(dataset_dir.rglob("*.parquet"))317    jsonl_files = sorted(dataset_dir.rglob("*.jsonl"))318    if not parquet_files and not jsonl_files:319        raise FileNotFoundError(f"No parquet/jsonl files found in {dataset_dir}")320 321    rows: list[dict[str, Any]] = []322    if parquet_files:323        import pandas as pd324 325        for parquet_file in parquet_files:326            frame = pd.read_parquet(parquet_file)327            rows.extend(frame.to_dict(orient="records"))328            if max_queries and len(rows) >= max_queries * 5:329                break330    else:331        for jsonl_file in jsonl_files:332            rows.extend(read_jsonl(jsonl_file))333            if max_queries and len(rows) >= max_queries * 5:334                break335 336    documents_by_id: dict[str, dict[str, Any]] = {}337    queries = []338    qrels: dict[str, set[str]] = {}339 340    for index, row in enumerate(rows):341        question = first_present(row, ["question", "query", "Question"])342        answer = first_present(row, ["answer", "Answer", "response"])343        context = first_present(row, ["context", "evidence", "gold_context", "text", "document"])344        table = first_present(row, ["table", "Table", "markdown_table"])345        doc_id = str(first_present(row, ["doc_id", "document_id", "filename", "pdf_path", "source"]) or f"row-{index}")346        if not question or not context:347            continue348 349        text = "\n".join(part for part in [str(context), str(table or "")] if part)350        if doc_id not in documents_by_id:351            documents_by_id[doc_id] = {352                "doc_id": doc_id,353                "title": str(first_present(row, ["company", "ticker", "title", "Title"]) or doc_id),354                "text": text,355                "metadata": {"source_dataset": "t2_ragbench", "answer": str(answer or "")},356            }357        queries.append(358            {359                "query_id": str(first_present(row, ["qid", "query_id", "id"]) or f"q-{index}"),360                "question": str(question),361                "relevant_doc_ids": [doc_id],362            }363        )364        qrels[queries[-1]["query_id"]] = {doc_id}365        if max_queries and len(queries) >= max_queries:366            break367 368    documents = list(documents_by_id.values())369    if max_corpus_docs:370        documents = documents[:max_corpus_docs]371        allowed = {document["doc_id"] for document in documents}372        queries = [query for query in queries if query["relevant_doc_ids"][0] in allowed]373        qrels = {query["query_id"]: set(query["relevant_doc_ids"]) for query in queries}374 375    if not documents or not queries:376        raise ValueError("T2-RAGBench produced no evaluable sample.")377 378    return EvalCorpus("t2_ragbench", documents, queries, qrels)379 380 381def first_present(row: dict[str, Any], keys: list[str]) -> Any:382    for key in keys:383        value = row.get(key)384        if value is not None and value != "":385            return value386    return None387 388 389def load_local_options_eval(max_queries: int | None) -> EvalCorpus:390    cases_path = EVAL_DIR / "local_options_eval.jsonl"391    if not cases_path.exists():392        raise FileNotFoundError(393            f"Local options eval set not found: {cases_path}. "394            "Create JSONL cases with question, expected_pages, expected_keywords."395        )396 397    from tools.query_knowledge import load_pdf_file398 399    pdf_files = sorted((PROJECT_ROOT / "knowledge_base" / "raw").rglob("*.pdf"))400    if not pdf_files:401        pdf_files = sorted((PROJECT_ROOT / "tools" / "knowledge_base" / "raw").rglob("*.pdf"))402    documents = []403    for pdf_file in pdf_files:404        for doc_index, document in enumerate(load_pdf_file(pdf_file)):405            documents.append(406                {407                    "doc_id": f"{pdf_file.name}:{document.metadata.get('page_number')}:{doc_index}",408                    "title": document.metadata.get("section_path") or pdf_file.name,409                    "text": document.text,410                    "metadata": document.metadata,411                }412            )413 414    queries = []415    qrels: dict[str, set[str]] = {}416    for case_index, case in enumerate(read_jsonl(cases_path)):417        query_id = str(case.get("id") or f"local-{case_index}")418        relevant_ids = []419        expected_pages = set(case.get("expected_pages") or [])420        expected_keywords = case.get("expected_keywords") or []421        for document in documents:422            metadata = document.get("metadata") or {}423            page_hit = metadata.get("page_number") in expected_pages424            keyword_hit = any(keyword in document["text"] for keyword in expected_keywords)425            if page_hit or keyword_hit:426                relevant_ids.append(document["doc_id"])427        queries.append(428            {429                "query_id": query_id,430                "question": case["question"],431                "relevant_doc_ids": relevant_ids,432            }433        )434        qrels[query_id] = set(relevant_ids)435        if max_queries and len(queries) >= max_queries:436            break437 438    if not documents or not queries:439        raise ValueError("Local options eval set produced no evaluable sample.")440 441    return EvalCorpus("local_options", documents, queries, qrels)442 443 444def load_eval_corpus(args: argparse.Namespace) -> EvalCorpus:445    dataset = DATASET_ALIASES.get(args.dataset, args.dataset)446    if dataset in {"scifact", "fiqa"}:447        return load_beir_dataset(dataset, args.split, args.max_corpus_docs, args.max_queries)448    if dataset == "open_ragbench":449        return load_open_ragbench(args.max_corpus_docs, args.max_queries)450    if dataset == "t2_ragbench":451        return load_t2_ragbench(args.max_corpus_docs, args.max_queries)452    if dataset == "local_options":453        return load_local_options_eval(args.max_queries)454    raise ValueError(f"Unknown dataset: {args.dataset}")455 456 457def collection_safe_name(value: str) -> str:458    safe = re.sub(r"[^A-Za-z0-9_-]+", "_", value)459    return safe.strip("_") or "default"460 461 462def build_index(corpus: EvalCorpus, chunk_size: int, chunk_overlap: int, rebuild: bool) -> VectorStoreIndex:463    configure_model_cache()464    from llama_index.embeddings.huggingface import HuggingFaceEmbedding465 466    index_path = INDEX_DIR / corpus.name467    if rebuild and index_path.exists():468        shutil.rmtree(index_path)469    index_path.mkdir(parents=True, exist_ok=True)470 471    db = chromadb.PersistentClient(path=str(index_path))472    embed_slug = collection_safe_name(EMBED_MODEL_NAME)473    collection_name = f"{corpus.name}_{embed_slug}_eval"474    if rebuild:475        try:476            db.delete_collection(collection_name)477        except Exception:478            pass479    collection = db.get_or_create_collection(collection_name)480    vector_store = ChromaVectorStore(chroma_collection=collection)481    storage_context = StorageContext.from_defaults(vector_store=vector_store)482    embed_model = HuggingFaceEmbedding(483        model_name=resolve_embed_model_name(),484        cache_folder=str(PROJECT_ROOT / "hf_cache" / "sentence_transformers"),485    )486 487    if collection.count() == 0:488        documents = [489            Document(490                text=document["text"],491                metadata={492                    "doc_id": document["doc_id"],493                    "title": document.get("title", ""),494                    **(document.get("metadata") or {}),495                },496            )497            for document in corpus.documents498        ]499        splitter = SentenceSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap)500        nodes = splitter.get_nodes_from_documents(documents)501        VectorStoreIndex(502            nodes,503            storage_context=storage_context,504            embed_model=embed_model,505            show_progress=True,506        )507 508    return VectorStoreIndex.from_vector_store(vector_store, embed_model=embed_model)509 510 511def build_bm25_retriever(corpus: EvalCorpus, chunk_size: int, chunk_overlap: int) -> BM25Retriever:512    documents = [513        Document(514            text=document["text"],515            metadata={516                "doc_id": document["doc_id"],517                "title": document.get("title", ""),518                **(document.get("metadata") or {}),519            },520        )521        for document in corpus.documents522    ]523    splitter = SentenceSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap)524    nodes = splitter.get_nodes_from_documents(documents)525    text_nodes = [526        TextNode(id_=node.node_id, text=node.get_content(), metadata=node.metadata)527        for node in nodes528    ]529    return BM25Retriever(text_nodes)530 531 532def merge_eval_results(533    vector_results: list[NodeWithScore],534    bm25_results: list[NodeWithScore],535    top_k: int,536) -> list[NodeWithScore]:537    merged: dict[str, NodeWithScore] = {}538 539    for rank, result in enumerate(vector_results):540        node_id = result.node.node_id541        merged[node_id] = NodeWithScore(node=result.node, score=1.0 / (rank + 1))542 543    for rank, result in enumerate(bm25_results):544        node_id = result.node.node_id545        reciprocal_rank_score = 1.0 / (rank + 1)546        if node_id in merged:547            merged[node_id].score = (merged[node_id].score or 0.0) + reciprocal_rank_score548        else:549            merged[node_id] = NodeWithScore(node=result.node, score=reciprocal_rank_score)550 551    results = list(merged.values())552    results.sort(key=lambda item: item.score or float("-inf"), reverse=True)553    return results[:top_k]554 555 556def evaluate_retrieval(557    corpus: EvalCorpus,558    index: VectorStoreIndex,559    top_k: int,560    use_reranker: bool = False,561    use_hybrid: bool = False,562    chunk_size: int = 512,563    chunk_overlap: int = 64,564    reranker_model_name: str = RERANKER_MODEL_NAME,565    reranker_candidates: int = 25,566) -> dict[str, Any]:567    retrieve_top_k = max(reranker_candidates, top_k) if use_reranker else max(top_k * 5, top_k)568    retriever = index.as_retriever(similarity_top_k=retrieve_top_k)569    bm25_retriever = (570        build_bm25_retriever(corpus, chunk_size, chunk_overlap)571        if use_hybrid572        else None573    )574    reranker = CrossEncoderReranker(reranker_model_name) if use_reranker else None575    cases = []576    hit_counts = {1: 0, 3: 0, 5: 0, top_k: 0}577    reciprocal_ranks = []578    ndcg_scores = []579 580    for query in corpus.queries:581        relevant_doc_ids = corpus.qrels.get(query["query_id"], set())582        vector_results = retriever.retrieve(query["question"])583        results = vector_results584        if bm25_retriever:585            bm25_results = bm25_retriever.retrieve(query["question"], retrieve_top_k)586            results = merge_eval_results(vector_results, bm25_results, retrieve_top_k)587        if reranker:588            results = reranker.rerank(589                query["question"],590                results,591                top_n=max(top_k * 5, top_k),592            )593        retrieved = []594        seen_doc_ids = set()595        first_hit_rank = None596        dcg = 0.0597 598        for result in results:599            metadata = result.node.metadata600            doc_id = str(metadata.get("doc_id", ""))601            if doc_id in seen_doc_ids:602                continue603            seen_doc_ids.add(doc_id)604            rank = len(retrieved) + 1605            hit = doc_id in relevant_doc_ids606            if hit and first_hit_rank is None:607                first_hit_rank = rank608            if hit:609                dcg += 1 / math.log2(rank + 1)610            retrieved.append(611                {612                    "rank": rank,613                    "doc_id": doc_id,614                    "score": result.score,615                    "hit": hit,616                    "title": metadata.get("title", ""),617                }618            )619            if len(retrieved) >= top_k:620                break621 622        ideal_hits = min(len(relevant_doc_ids), top_k)623        idcg = sum(1 / math.log2(rank + 1) for rank in range(1, ideal_hits + 1))624        ndcg = dcg / idcg if idcg else 0.0625        ndcg_scores.append(ndcg)626        reciprocal_ranks.append(1 / first_hit_rank if first_hit_rank else 0.0)627 628        for k in hit_counts:629            if any(item["hit"] for item in retrieved[:k]):630                hit_counts[k] += 1631 632        cases.append(633            {634                "query_id": query["query_id"],635                "question": query["question"],636                "relevant_doc_ids": sorted(relevant_doc_ids),637                "first_hit_rank": first_hit_rank,638                "retrieved": retrieved,639            }640        )641 642    total = len(corpus.queries)643    metrics = {644        "queries": total,645        "documents": len(corpus.documents),646        "top_k": top_k,647        "mrr": sum(reciprocal_ranks) / total if total else 0.0,648        "ndcg_at_k": sum(ndcg_scores) / total if total else 0.0,649        "reranker_enabled": use_reranker,650        "hybrid_enabled": use_hybrid,651    }652    for k, count in sorted(hit_counts.items()):653        metrics[f"hit_at_{k}"] = count / total if total else 0.0654 655    return {"dataset": corpus.name, "metrics": metrics, "cases": cases}656 657 658def write_reports(report: dict[str, Any]) -> tuple[Path, Path]:659    ensure_dirs()660    dataset_name = report["dataset"]661    json_path = REPORT_DIR / f"{dataset_name}_retrieval_eval.json"662    md_path = REPORT_DIR / f"{dataset_name}_retrieval_eval.md"663    json_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")664 665    metrics = report["metrics"]666    lines = [667        f"# Retrieval Eval: {dataset_name}",668        "",669        "## Metrics",670        "",671    ]672    for key, value in metrics.items():673        lines.append(f"- `{key}`: {value:.4f}" if isinstance(value, float) else f"- `{key}`: {value}")674 675    lines.extend(["", "## Sample Cases", ""])676    for case in report["cases"][:10]:677        lines.append(f"### {case['query_id']}")678        lines.append("")679        lines.append(case["question"])680        lines.append("")681        lines.append(f"- first_hit_rank: `{case['first_hit_rank']}`")682        for item in case["retrieved"][:5]:683            lines.append(684                f"- rank {item['rank']}: hit={item['hit']} doc_id=`{item['doc_id']}` score={item['score']}"685            )686        lines.append("")687 688    md_path.write_text("\n".join(lines), encoding="utf-8")689    return json_path, md_path690 691 692def parse_args() -> argparse.Namespace:693    parser = argparse.ArgumentParser(description="Run retrieval eval for RAG datasets.")694    parser.add_argument(695        "--dataset",696        required=True,697        help="beir/scifact, beir/fiqa, open-ragbench, t2-ragbench, or local-options",698    )699    parser.add_argument("--split", default="test")700    parser.add_argument("--top-k", type=int, default=5)701    parser.add_argument("--chunk-size", type=int, default=512)702    parser.add_argument("--chunk-overlap", type=int, default=64)703    parser.add_argument("--max-corpus-docs", type=int, default=None)704    parser.add_argument("--max-queries", type=int, default=None)705    parser.add_argument("--rebuild", action="store_true")706    parser.add_argument("--use-hybrid", action="store_true")707    parser.add_argument("--use-reranker", action="store_true")708    parser.add_argument("--reranker-model", default=RERANKER_MODEL_NAME)709    parser.add_argument("--reranker-candidates", type=int, default=25)710    return parser.parse_args()711 712 713def main() -> None:714    args = parse_args()715    corpus = load_eval_corpus(args)716    index = build_index(corpus, args.chunk_size, args.chunk_overlap, args.rebuild)717    report = evaluate_retrieval(718        corpus,719        index,720        args.top_k,721        use_reranker=args.use_reranker,722        use_hybrid=args.use_hybrid,723        chunk_size=args.chunk_size,724        chunk_overlap=args.chunk_overlap,725        reranker_model_name=args.reranker_model,726        reranker_candidates=args.reranker_candidates,727    )728    json_path, md_path = write_reports(report)729    print(json.dumps(report["metrics"], ensure_ascii=False, indent=2))730    print(f"JSON report: {json_path}")731    print(f"Markdown report: {md_path}")732 733 734if __name__ == "__main__":735    main()736