Team Ai
Datasetpublic

Angshul/SparseGeometricRAG

SparseGeometricRAG CPU-first sparse geometric retrieval for practical top-10 RAG No transformer inference at retrieval time. No retrieval GPU requirement. No dense document-vector dot products. No external API. SparseGeometricRAG is a retrieval system built around one systems objective: make the retrieval layer cheap enough to run on ordinary multicore CPU hardware without turning the corpus into a dense embedding database. It uses sparse TF-IDF geometry, fuzzy… See the full description on the dataset page: https://huggingface.co/datasets/Angshul/SparseGeometricRAG.

sourceHugging Facefair-noncommercial-research-licenseupdated 2mo agoView on Hugging Face
0likes331downloads
msmarco_dev_fast.py28 linesDownload Raw Back to msmarco_scale
1import sys,time,json,numpy as np,pandas as pd2sys.path.insert(0,'/mnt/data')3from msmarco_full_search_fast import FullIndex,load_query_texts,qrels_from_tsv,eval_run,ROOT,WORK,P4BEST_H=05idx=FullIndex(); print('loaded',idx.meta,flush=True)6df=pd.read_csv(ROOT/'dev.tsv',sep='\t',usecols=['query-id']); ids=[str(x) for x in np.unique(df['query-id'].to_numpy())]; del df7texts=load_query_texts(ids); qrels=qrels_from_tsv(ROOT/'dev.tsv',ids,positive_only=True)8# warm compile/pages, not timed9w=idx.prepare(texts[ids[0]],hmax=1); idx.rank_h(w,0,100); del w10run={}; times=[]; cands=[]; mems=[]11route_num=pool_num=rel_den=012for z,qid in enumerate(ids):13 t=time.perf_counter(); pp=idx.prepare(texts[qid],hmax=1); rank=idx.rank_h(pp,BEST_H,100); dt=(time.perf_counter()-t)*100014 run[qid]=rank; times.append(dt); cands.append(pp['candidate_docs'] if pp else 0); mems.append(pp['candidate_memberships'] if pp else 0)15 rel=[int(d) for d,r in qrels[qid].items() if r>0]; rel_den+=len(rel)16 if pp is not None:17  ud=pp['ud']; pool=pp['cand_docs'][:P]18  for d in rel:19   k=np.searchsorted(ud,d); route_num += int(k<len(ud) and int(ud[k])==d); pool_num += int(np.any(pool==d))20 if (z+1)%500==0:21  print('dev',z+1,'median_ms',float(np.median(times)),'p95_ms',float(np.percentile(times,95)),'avgcand',float(np.mean(cands)),'route_rel',route_num/rel_den,'pool_rel',pool_num/rel_den,flush=True)22m=eval_run(run,qrels)23timing={'median_ms':float(np.median(times)),'p95_ms':float(np.percentile(times,95)),'mean_ms':float(np.mean(times)),'qps':1000/float(np.mean(times)),'avg_candidate_docs':float(np.mean(cands)),'median_candidate_docs':float(np.median(cands)),'avg_candidate_memberships':float(np.mean(mems))}24stages={'routed_relevant_recall':route_num/rel_den,'pool_P2000_relevant_recall':pool_num/rel_den,'final_R100':m['R@100'],'total_positive_qrels':rel_den}25out={'protocol':'h=0 selected on deterministic 1000-query TRAIN validation sweep; full 6980-query DEV untouched','best_h':0,'dev_metrics':m,'timing':timing,'stage_diagnostics':stages,'index_meta':idx.meta,'geometry_note':'full 8.84M vocabulary/IDF; geometric codebook calibrated on first 1M passages','physical_optimizations':'branch-sorted postings; query-local dense center/reliability lookup; global compact support CSR; ranking verified identical on regression queries'}26with open(WORK/'full_msmarco_dev_fast_results.json','w') as f: json.dump(out,f,indent=2)27print('DEV_METRICS',m,flush=True); print('TIMING',timing,flush=True); print('STAGES',stages,flush=True); print('saved',WORK/'full_msmarco_dev_fast_results.json',flush=True)28