mathidot111/First_agent_template
0
1from __future__ import annotations2 3import argparse4import csv5import json6import math7import re8import shutil9import zipfile10from dataclasses import dataclass11from pathlib import Path12from typing import Any, Iterable13 14import chromadb15import requests16from llama_index.core import StorageContext, VectorStoreIndex17from llama_index.core.node_parser import SentenceSplitter18from llama_index.core.schema import Document19from llama_index.core.schema import NodeWithScore, TextNode20from llama_index.vector_stores.chroma import ChromaVectorStore21 22from tools.query_knowledge import (23 BM25Retriever,24 EMBED_MODEL_NAME,25 RERANKER_MODEL_NAME,26 CrossEncoderReranker,27 configure_model_cache,28 resolve_embed_model_name,29)30 31 32PROJECT_ROOT = Path(__file__).resolve().parents[1]33EVAL_DIR = PROJECT_ROOT / "eval"34DATA_DIR = EVAL_DIR / "data"35INDEX_DIR = EVAL_DIR / "indexes"36REPORT_DIR = EVAL_DIR / "reports"37 38BEIR_URLS = {39 "scifact": "https://public.ukp.informatik.tu-darmstadt.de/thakur/BEIR/datasets/scifact.zip",40 "fiqa": "https://public.ukp.informatik.tu-darmstadt.de/thakur/BEIR/datasets/fiqa.zip",41}42 43DATASET_ALIASES = {44 "beir/scifact": "scifact",45 "beir/fiqa": "fiqa",46 "open-ragbench": "open_ragbench",47 "open_ragbench": "open_ragbench",48 "t2-ragbench": "t2_ragbench",49 "t2_ragbench": "t2_ragbench",50 "local-options": "local_options",51 "local_options": "local_options",52}53 54 55@dataclass56class EvalCorpus:57 name: str58 documents: list[dict[str, Any]]59 queries: list[dict[str, Any]]60 qrels: dict[str, set[str]]61 62 63def ensure_dirs() -> None:64 DATA_DIR.mkdir(parents=True, exist_ok=True)65 INDEX_DIR.mkdir(parents=True, exist_ok=True)66 REPORT_DIR.mkdir(parents=True, exist_ok=True)67 68 69def download_file(url: str, destination: Path) -> None:70 destination.parent.mkdir(parents=True, exist_ok=True)71 with requests.get(url, stream=True, timeout=60) as response:72 response.raise_for_status()73 with destination.open("wb") as file:74 for chunk in response.iter_content(chunk_size=1024 * 1024):75 if chunk:76 file.write(chunk)77 78 79def read_jsonl(path: Path) -> Iterable[dict[str, Any]]:80 with path.open("r", encoding="utf-8") as file:81 for line in file:82 line = line.strip()83 if line:84 yield json.loads(line)85 86 87def prepare_beir_dataset(dataset_name: str) -> Path:88 ensure_dirs()89 if dataset_name not in BEIR_URLS:90 raise ValueError(f"Unsupported BEIR dataset: {dataset_name}")91 92 target_dir = DATA_DIR / "beir" / dataset_name93 corpus_path = target_dir / "corpus.jsonl"94 if corpus_path.exists():95 return target_dir96 97 zip_path = DATA_DIR / "downloads" / f"{dataset_name}.zip"98 if not zip_path.exists():99 download_file(BEIR_URLS[dataset_name], zip_path)100 101 extract_root = DATA_DIR / "beir"102 extract_root.mkdir(parents=True, exist_ok=True)103 with zipfile.ZipFile(zip_path) as archive:104 archive.extractall(extract_root)105 106 if not corpus_path.exists():107 raise FileNotFoundError(f"BEIR extraction did not create {corpus_path}")108 109 return target_dir110 111 112def load_beir_dataset(113 dataset_name: str,114 split: str,115 max_corpus_docs: int | None,116 max_queries: int | None,117) -> EvalCorpus:118 dataset_dir = prepare_beir_dataset(dataset_name)119 120 all_queries = {121 str(row["_id"]): row.get("text", "")122 for row in read_jsonl(dataset_dir / "queries.jsonl")123 }124 125 qrels_path = dataset_dir / "qrels" / f"{split}.tsv"126 if not qrels_path.exists():127 candidates = sorted((dataset_dir / "qrels").glob("*.tsv"))128 if not candidates:129 raise FileNotFoundError(f"No qrels found under {dataset_dir / 'qrels'}")130 qrels_path = candidates[0]131 132 all_qrels: dict[str, set[str]] = {}133 with qrels_path.open("r", encoding="utf-8") as file:134 reader = csv.DictReader(file, delimiter="\t")135 for row in reader:136 query_id = str(row.get("query-id") or row.get("query_id"))137 corpus_id = str(row.get("corpus-id") or row.get("corpus_id"))138 score = int(row.get("score", 1))139 if score <= 0:140 continue141 all_qrels.setdefault(query_id, set()).add(corpus_id)142 143 queries = []144 required_doc_ids = set()145 for query_id, relevant_docs in all_qrels.items():146 if query_id not in all_queries:147 continue148 if max_corpus_docs and len(required_doc_ids | relevant_docs) > max_corpus_docs:149 continue150 required_doc_ids.update(relevant_docs)151 queries.append(152 {153 "query_id": query_id,154 "question": all_queries[query_id],155 "relevant_doc_ids": sorted(relevant_docs),156 }157 )158 if max_queries and len(queries) >= max_queries:159 break160 161 documents = []162 seen_doc_ids = set()163 for row in read_jsonl(dataset_dir / "corpus.jsonl"):164 doc_id = str(row["_id"])165 if required_doc_ids and doc_id not in required_doc_ids:166 if max_corpus_docs and len(documents) >= max_corpus_docs:167 continue168 if max_corpus_docs and len(documents) + len(required_doc_ids - seen_doc_ids) >= max_corpus_docs:169 continue170 title = row.get("title") or ""171 text = row.get("text") or ""172 documents.append(173 {174 "doc_id": doc_id,175 "title": title,176 "text": f"{title}\n{text}".strip(),177 "metadata": {"source_dataset": f"beir/{dataset_name}"},178 }179 )180 seen_doc_ids.add(doc_id)181 if max_corpus_docs and len(documents) >= max_corpus_docs and required_doc_ids.issubset(seen_doc_ids):182 break183 184 if not documents or not queries:185 raise ValueError(186 f"Dataset beir/{dataset_name} has no evaluable documents/queries. "187 "Increase --max-corpus-docs or use a larger sample."188 )189 190 return EvalCorpus(191 name=f"beir_{dataset_name}",192 documents=documents,193 queries=queries,194 qrels={query["query_id"]: set(query["relevant_doc_ids"]) for query in queries},195 )196 197 198def snapshot_hf_dataset(repo_id: str, local_name: str) -> Path:199 from huggingface_hub import snapshot_download200 201 ensure_dirs()202 target_dir = DATA_DIR / "hf" / local_name203 if target_dir.exists():204 return target_dir205 206 snapshot_download(207 repo_id=repo_id,208 repo_type="dataset",209 local_dir=str(target_dir),210 local_dir_use_symlinks=False,211 )212 return target_dir213 214 215def flatten_open_ragbench_section(section: dict[str, Any]) -> str:216 parts = [section.get("text") or ""]217 tables = section.get("tables") or {}218 if isinstance(tables, dict):219 parts.extend(str(value) for value in tables.values())220 return "\n".join(part for part in parts if part)221 222 223def load_open_ragbench(224 max_corpus_docs: int | None,225 max_queries: int | None,226) -> EvalCorpus:227 dataset_dir = snapshot_hf_dataset("vectara/open_ragbench", "open_ragbench")228 root = dataset_dir / "pdf" / "arxiv"229 if not root.exists():230 root = dataset_dir / "official" / "pdf" / "arxiv"231 if not root.exists():232 raise FileNotFoundError(f"Open RAGBench root not found: {root}")233 234 queries_data = json.loads((root / "queries.json").read_text(encoding="utf-8"))235 qrels_data = json.loads((root / "qrels.json").read_text(encoding="utf-8"))236 237 documents = []238 qrels: dict[str, set[str]] = {}239 required_doc_ids = set()240 selected_query_ids = []241 for query_id, qrel in qrels_data.items():242 doc_id = str(qrel.get("doc_id"))243 if not doc_id or doc_id == "None":244 continue245 selected_query_ids.append(str(query_id))246 required_doc_ids.add(doc_id)247 if max_queries and len(selected_query_ids) >= max_queries:248 break249 250 allowed_doc_ids = set()251 corpus_files = sorted((root / "corpus").glob("*.json"))252 253 for corpus_file in corpus_files:254 paper = json.loads(corpus_file.read_text(encoding="utf-8"))255 paper_id = str(paper.get("id") or corpus_file.stem)256 is_required = paper_id in required_doc_ids257 if max_corpus_docs and not is_required:258 missing_required_count = len(required_doc_ids - allowed_doc_ids)259 if len(documents) + missing_required_count >= max_corpus_docs:260 continue261 allowed_doc_ids.add(paper_id)262 section_texts = []263 for section_index, section in enumerate(paper.get("sections") or []):264 section_text = flatten_open_ragbench_section(section)265 if section_text:266 section_texts.append(f"[section {section_index}]\n{section_text}")267 text = "\n\n".join(268 part269 for part in [paper.get("title") or "", paper.get("abstract") or "", *section_texts]270 if part271 )272 documents.append(273 {274 "doc_id": paper_id,275 "title": paper.get("title") or paper_id,276 "text": text,277 "metadata": {278 "source_dataset": "open_ragbench",279 "categories": ",".join(paper.get("categories") or []),280 },281 }282 )283 if max_corpus_docs and len(documents) >= max_corpus_docs:284 break285 286 queries = []287 for query_id in selected_query_ids:288 qrel = qrels_data[query_id]289 doc_id = str(qrel.get("doc_id"))290 if doc_id not in allowed_doc_ids:291 continue292 query_payload = queries_data.get(query_id) or {}293 question = query_payload.get("query") if isinstance(query_payload, dict) else str(query_payload)294 qrels[str(query_id)] = {doc_id}295 queries.append(296 {297 "query_id": str(query_id),298 "question": question,299 "relevant_doc_ids": [doc_id],300 }301 )302 if max_queries and len(queries) >= max_queries:303 break304 305 if not documents or not queries:306 raise ValueError("Open RAGBench produced no evaluable sample.")307 308 return EvalCorpus("open_ragbench", documents, queries, qrels)309 310 311def load_t2_ragbench(312 max_corpus_docs: int | None,313 max_queries: int | None,314) -> EvalCorpus:315 dataset_dir = snapshot_hf_dataset("G4KMU/t2-ragbench", "t2_ragbench")316 parquet_files = sorted(dataset_dir.rglob("*.parquet"))317 jsonl_files = sorted(dataset_dir.rglob("*.jsonl"))318 if not parquet_files and not jsonl_files:319 raise FileNotFoundError(f"No parquet/jsonl files found in {dataset_dir}")320 321 rows: list[dict[str, Any]] = []322 if parquet_files:323 import pandas as pd324 325 for parquet_file in parquet_files:326 frame = pd.read_parquet(parquet_file)327 rows.extend(frame.to_dict(orient="records"))328 if max_queries and len(rows) >= max_queries * 5:329 break330 else:331 for jsonl_file in jsonl_files:332 rows.extend(read_jsonl(jsonl_file))333 if max_queries and len(rows) >= max_queries * 5:334 break335 336 documents_by_id: dict[str, dict[str, Any]] = {}337 queries = []338 qrels: dict[str, set[str]] = {}339 340 for index, row in enumerate(rows):341 question = first_present(row, ["question", "query", "Question"])342 answer = first_present(row, ["answer", "Answer", "response"])343 context = first_present(row, ["context", "evidence", "gold_context", "text", "document"])344 table = first_present(row, ["table", "Table", "markdown_table"])345 doc_id = str(first_present(row, ["doc_id", "document_id", "filename", "pdf_path", "source"]) or f"row-{index}")346 if not question or not context:347 continue348 349 text = "\n".join(part for part in [str(context), str(table or "")] if part)350 if doc_id not in documents_by_id:351 documents_by_id[doc_id] = {352 "doc_id": doc_id,353 "title": str(first_present(row, ["company", "ticker", "title", "Title"]) or doc_id),354 "text": text,355 "metadata": {"source_dataset": "t2_ragbench", "answer": str(answer or "")},356 }357 queries.append(358 {359 "query_id": str(first_present(row, ["qid", "query_id", "id"]) or f"q-{index}"),360 "question": str(question),361 "relevant_doc_ids": [doc_id],362 }363 )364 qrels[queries[-1]["query_id"]] = {doc_id}365 if max_queries and len(queries) >= max_queries:366 break367 368 documents = list(documents_by_id.values())369 if max_corpus_docs:370 documents = documents[:max_corpus_docs]371 allowed = {document["doc_id"] for document in documents}372 queries = [query for query in queries if query["relevant_doc_ids"][0] in allowed]373 qrels = {query["query_id"]: set(query["relevant_doc_ids"]) for query in queries}374 375 if not documents or not queries:376 raise ValueError("T2-RAGBench produced no evaluable sample.")377 378 return EvalCorpus("t2_ragbench", documents, queries, qrels)379 380 381def first_present(row: dict[str, Any], keys: list[str]) -> Any:382 for key in keys:383 value = row.get(key)384 if value is not None and value != "":385 return value386 return None387 388 389def load_local_options_eval(max_queries: int | None) -> EvalCorpus:390 cases_path = EVAL_DIR / "local_options_eval.jsonl"391 if not cases_path.exists():392 raise FileNotFoundError(393 f"Local options eval set not found: {cases_path}. "394 "Create JSONL cases with question, expected_pages, expected_keywords."395 )396 397 from tools.query_knowledge import load_pdf_file398 399 pdf_files = sorted((PROJECT_ROOT / "knowledge_base" / "raw").rglob("*.pdf"))400 if not pdf_files:401 pdf_files = sorted((PROJECT_ROOT / "tools" / "knowledge_base" / "raw").rglob("*.pdf"))402 documents = []403 for pdf_file in pdf_files:404 for doc_index, document in enumerate(load_pdf_file(pdf_file)):405 documents.append(406 {407 "doc_id": f"{pdf_file.name}:{document.metadata.get('page_number')}:{doc_index}",408 "title": document.metadata.get("section_path") or pdf_file.name,409 "text": document.text,410 "metadata": document.metadata,411 }412 )413 414 queries = []415 qrels: dict[str, set[str]] = {}416 for case_index, case in enumerate(read_jsonl(cases_path)):417 query_id = str(case.get("id") or f"local-{case_index}")418 relevant_ids = []419 expected_pages = set(case.get("expected_pages") or [])420 expected_keywords = case.get("expected_keywords") or []421 for document in documents:422 metadata = document.get("metadata") or {}423 page_hit = metadata.get("page_number") in expected_pages424 keyword_hit = any(keyword in document["text"] for keyword in expected_keywords)425 if page_hit or keyword_hit:426 relevant_ids.append(document["doc_id"])427 queries.append(428 {429 "query_id": query_id,430 "question": case["question"],431 "relevant_doc_ids": relevant_ids,432 }433 )434 qrels[query_id] = set(relevant_ids)435 if max_queries and len(queries) >= max_queries:436 break437 438 if not documents or not queries:439 raise ValueError("Local options eval set produced no evaluable sample.")440 441 return EvalCorpus("local_options", documents, queries, qrels)442 443 444def load_eval_corpus(args: argparse.Namespace) -> EvalCorpus:445 dataset = DATASET_ALIASES.get(args.dataset, args.dataset)446 if dataset in {"scifact", "fiqa"}:447 return load_beir_dataset(dataset, args.split, args.max_corpus_docs, args.max_queries)448 if dataset == "open_ragbench":449 return load_open_ragbench(args.max_corpus_docs, args.max_queries)450 if dataset == "t2_ragbench":451 return load_t2_ragbench(args.max_corpus_docs, args.max_queries)452 if dataset == "local_options":453 return load_local_options_eval(args.max_queries)454 raise ValueError(f"Unknown dataset: {args.dataset}")455 456 457def collection_safe_name(value: str) -> str:458 safe = re.sub(r"[^A-Za-z0-9_-]+", "_", value)459 return safe.strip("_") or "default"460 461 462def build_index(corpus: EvalCorpus, chunk_size: int, chunk_overlap: int, rebuild: bool) -> VectorStoreIndex:463 configure_model_cache()464 from llama_index.embeddings.huggingface import HuggingFaceEmbedding465 466 index_path = INDEX_DIR / corpus.name467 if rebuild and index_path.exists():468 shutil.rmtree(index_path)469 index_path.mkdir(parents=True, exist_ok=True)470 471 db = chromadb.PersistentClient(path=str(index_path))472 embed_slug = collection_safe_name(EMBED_MODEL_NAME)473 collection_name = f"{corpus.name}_{embed_slug}_eval"474 if rebuild:475 try:476 db.delete_collection(collection_name)477 except Exception:478 pass479 collection = db.get_or_create_collection(collection_name)480 vector_store = ChromaVectorStore(chroma_collection=collection)481 storage_context = StorageContext.from_defaults(vector_store=vector_store)482 embed_model = HuggingFaceEmbedding(483 model_name=resolve_embed_model_name(),484 cache_folder=str(PROJECT_ROOT / "hf_cache" / "sentence_transformers"),485 )486 487 if collection.count() == 0:488 documents = [489 Document(490 text=document["text"],491 metadata={492 "doc_id": document["doc_id"],493 "title": document.get("title", ""),494 **(document.get("metadata") or {}),495 },496 )497 for document in corpus.documents498 ]499 splitter = SentenceSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap)500 nodes = splitter.get_nodes_from_documents(documents)501 VectorStoreIndex(502 nodes,503 storage_context=storage_context,504 embed_model=embed_model,505 show_progress=True,506 )507 508 return VectorStoreIndex.from_vector_store(vector_store, embed_model=embed_model)509 510 511def build_bm25_retriever(corpus: EvalCorpus, chunk_size: int, chunk_overlap: int) -> BM25Retriever:512 documents = [513 Document(514 text=document["text"],515 metadata={516 "doc_id": document["doc_id"],517 "title": document.get("title", ""),518 **(document.get("metadata") or {}),519 },520 )521 for document in corpus.documents522 ]523 splitter = SentenceSplitter(chunk_size=chunk_size, chunk_overlap=chunk_overlap)524 nodes = splitter.get_nodes_from_documents(documents)525 text_nodes = [526 TextNode(id_=node.node_id, text=node.get_content(), metadata=node.metadata)527 for node in nodes528 ]529 return BM25Retriever(text_nodes)530 531 532def merge_eval_results(533 vector_results: list[NodeWithScore],534 bm25_results: list[NodeWithScore],535 top_k: int,536) -> list[NodeWithScore]:537 merged: dict[str, NodeWithScore] = {}538 539 for rank, result in enumerate(vector_results):540 node_id = result.node.node_id541 merged[node_id] = NodeWithScore(node=result.node, score=1.0 / (rank + 1))542 543 for rank, result in enumerate(bm25_results):544 node_id = result.node.node_id545 reciprocal_rank_score = 1.0 / (rank + 1)546 if node_id in merged:547 merged[node_id].score = (merged[node_id].score or 0.0) + reciprocal_rank_score548 else:549 merged[node_id] = NodeWithScore(node=result.node, score=reciprocal_rank_score)550 551 results = list(merged.values())552 results.sort(key=lambda item: item.score or float("-inf"), reverse=True)553 return results[:top_k]554 555 556def evaluate_retrieval(557 corpus: EvalCorpus,558 index: VectorStoreIndex,559 top_k: int,560 use_reranker: bool = False,561 use_hybrid: bool = False,562 chunk_size: int = 512,563 chunk_overlap: int = 64,564 reranker_model_name: str = RERANKER_MODEL_NAME,565 reranker_candidates: int = 25,566) -> dict[str, Any]:567 retrieve_top_k = max(reranker_candidates, top_k) if use_reranker else max(top_k * 5, top_k)568 retriever = index.as_retriever(similarity_top_k=retrieve_top_k)569 bm25_retriever = (570 build_bm25_retriever(corpus, chunk_size, chunk_overlap)571 if use_hybrid572 else None573 )574 reranker = CrossEncoderReranker(reranker_model_name) if use_reranker else None575 cases = []576 hit_counts = {1: 0, 3: 0, 5: 0, top_k: 0}577 reciprocal_ranks = []578 ndcg_scores = []579 580 for query in corpus.queries:581 relevant_doc_ids = corpus.qrels.get(query["query_id"], set())582 vector_results = retriever.retrieve(query["question"])583 results = vector_results584 if bm25_retriever:585 bm25_results = bm25_retriever.retrieve(query["question"], retrieve_top_k)586 results = merge_eval_results(vector_results, bm25_results, retrieve_top_k)587 if reranker:588 results = reranker.rerank(589 query["question"],590 results,591 top_n=max(top_k * 5, top_k),592 )593 retrieved = []594 seen_doc_ids = set()595 first_hit_rank = None596 dcg = 0.0597 598 for result in results:599 metadata = result.node.metadata600 doc_id = str(metadata.get("doc_id", ""))601 if doc_id in seen_doc_ids:602 continue603 seen_doc_ids.add(doc_id)604 rank = len(retrieved) + 1605 hit = doc_id in relevant_doc_ids606 if hit and first_hit_rank is None:607 first_hit_rank = rank608 if hit:609 dcg += 1 / math.log2(rank + 1)610 retrieved.append(611 {612 "rank": rank,613 "doc_id": doc_id,614 "score": result.score,615 "hit": hit,616 "title": metadata.get("title", ""),617 }618 )619 if len(retrieved) >= top_k:620 break621 622 ideal_hits = min(len(relevant_doc_ids), top_k)623 idcg = sum(1 / math.log2(rank + 1) for rank in range(1, ideal_hits + 1))624 ndcg = dcg / idcg if idcg else 0.0625 ndcg_scores.append(ndcg)626 reciprocal_ranks.append(1 / first_hit_rank if first_hit_rank else 0.0)627 628 for k in hit_counts:629 if any(item["hit"] for item in retrieved[:k]):630 hit_counts[k] += 1631 632 cases.append(633 {634 "query_id": query["query_id"],635 "question": query["question"],636 "relevant_doc_ids": sorted(relevant_doc_ids),637 "first_hit_rank": first_hit_rank,638 "retrieved": retrieved,639 }640 )641 642 total = len(corpus.queries)643 metrics = {644 "queries": total,645 "documents": len(corpus.documents),646 "top_k": top_k,647 "mrr": sum(reciprocal_ranks) / total if total else 0.0,648 "ndcg_at_k": sum(ndcg_scores) / total if total else 0.0,649 "reranker_enabled": use_reranker,650 "hybrid_enabled": use_hybrid,651 }652 for k, count in sorted(hit_counts.items()):653 metrics[f"hit_at_{k}"] = count / total if total else 0.0654 655 return {"dataset": corpus.name, "metrics": metrics, "cases": cases}656 657 658def write_reports(report: dict[str, Any]) -> tuple[Path, Path]:659 ensure_dirs()660 dataset_name = report["dataset"]661 json_path = REPORT_DIR / f"{dataset_name}_retrieval_eval.json"662 md_path = REPORT_DIR / f"{dataset_name}_retrieval_eval.md"663 json_path.write_text(json.dumps(report, ensure_ascii=False, indent=2), encoding="utf-8")664 665 metrics = report["metrics"]666 lines = [667 f"# Retrieval Eval: {dataset_name}",668 "",669 "## Metrics",670 "",671 ]672 for key, value in metrics.items():673 lines.append(f"- `{key}`: {value:.4f}" if isinstance(value, float) else f"- `{key}`: {value}")674 675 lines.extend(["", "## Sample Cases", ""])676 for case in report["cases"][:10]:677 lines.append(f"### {case['query_id']}")678 lines.append("")679 lines.append(case["question"])680 lines.append("")681 lines.append(f"- first_hit_rank: `{case['first_hit_rank']}`")682 for item in case["retrieved"][:5]:683 lines.append(684 f"- rank {item['rank']}: hit={item['hit']} doc_id=`{item['doc_id']}` score={item['score']}"685 )686 lines.append("")687 688 md_path.write_text("\n".join(lines), encoding="utf-8")689 return json_path, md_path690 691 692def parse_args() -> argparse.Namespace:693 parser = argparse.ArgumentParser(description="Run retrieval eval for RAG datasets.")694 parser.add_argument(695 "--dataset",696 required=True,697 help="beir/scifact, beir/fiqa, open-ragbench, t2-ragbench, or local-options",698 )699 parser.add_argument("--split", default="test")700 parser.add_argument("--top-k", type=int, default=5)701 parser.add_argument("--chunk-size", type=int, default=512)702 parser.add_argument("--chunk-overlap", type=int, default=64)703 parser.add_argument("--max-corpus-docs", type=int, default=None)704 parser.add_argument("--max-queries", type=int, default=None)705 parser.add_argument("--rebuild", action="store_true")706 parser.add_argument("--use-hybrid", action="store_true")707 parser.add_argument("--use-reranker", action="store_true")708 parser.add_argument("--reranker-model", default=RERANKER_MODEL_NAME)709 parser.add_argument("--reranker-candidates", type=int, default=25)710 return parser.parse_args()711 712 713def main() -> None:714 args = parse_args()715 corpus = load_eval_corpus(args)716 index = build_index(corpus, args.chunk_size, args.chunk_overlap, args.rebuild)717 report = evaluate_retrieval(718 corpus,719 index,720 args.top_k,721 use_reranker=args.use_reranker,722 use_hybrid=args.use_hybrid,723 chunk_size=args.chunk_size,724 chunk_overlap=args.chunk_overlap,725 reranker_model_name=args.reranker_model,726 reranker_candidates=args.reranker_candidates,727 )728 json_path, md_path = write_reports(report)729 print(json.dumps(report["metrics"], ensure_ascii=False, indent=2))730 print(f"JSON report: {json_path}")731 print(f"Markdown report: {md_path}")732 733 734if __name__ == "__main__":735 main()736 