Team Ai
Datasetpublic

mteb/CodeSearchNetCCRetrieval

CodeSearchNetCCRetrieval An MTEB dataset Massive Text Embedding Benchmark The dataset is a collection of code snippets. The task is to retrieve the most relevant code snippet for a given code snippet. Task category t2t Domains Programming, Written Reference https://arxiv.org/abs/2407.02883 Source datasets: CoIR-Retrieval/CodeSearchNet-ccr How to evaluate on this task You can evaluate an embedding model on this dataset using the following code:… See the full description on the dataset page: https://huggingface.co/datasets/mteb/CodeSearchNetCCRetrieval.

sourceHugging Facemitupdated 1y agoView on Hugging Face
0likes185downloads
README.md626 linesDownload Raw Back to root
1---2annotations_creators:3- derived4language:5- code6license: mit7multilinguality: monolingual8source_datasets:9- CoIR-Retrieval/CodeSearchNet-ccr10task_categories:11- text-retrieval12task_ids: []13dataset_info:14- config_name: go-corpus15  features:16  - name: id17    dtype: string18  - name: text19    dtype: string20  - name: title21    dtype: string22  splits:23  - name: test24    num_bytes: 3480983925    num_examples: 18273526  download_size: 1572590927  dataset_size: 3480983928- config_name: go-qrels29  features:30  - name: query-id31    dtype: string32  - name: corpus-id33    dtype: string34  - name: score35    dtype: int6436  splits:37  - name: test38    num_bytes: 24366039    num_examples: 812240  download_size: 9974141  dataset_size: 24366042- config_name: go-queries43  features:44  - name: id45    dtype: string46  - name: text47    dtype: string48  splits:49  - name: test50    num_bytes: 202082451    num_examples: 812252  download_size: 91125553  dataset_size: 202082454- config_name: java-corpus55  features:56  - name: id57    dtype: string58  - name: text59    dtype: string60  - name: title61    dtype: string62  splits:63  - name: test64    num_bytes: 4902701865    num_examples: 18106166  download_size: 1944100467  dataset_size: 4902701868- config_name: java-qrels69  features:70  - name: query-id71    dtype: string72  - name: corpus-id73    dtype: string74  - name: score75    dtype: int6476  splits:77  - name: test78    num_bytes: 32865079    num_examples: 1095580  download_size: 13707781  dataset_size: 32865082- config_name: java-queries83  features:84  - name: id85    dtype: string86  - name: text87    dtype: string88  splits:89  - name: test90    num_bytes: 391692191    num_examples: 1095592  download_size: 158968393  dataset_size: 391692194- config_name: javascript-corpus95  features:96  - name: id97    dtype: string98  - name: text99    dtype: string100  - name: title101    dtype: string102  splits:103  - name: test104    num_bytes: 18616585105    num_examples: 65201106  download_size: 8405911107  dataset_size: 18616585108- config_name: javascript-qrels109  features:110  - name: query-id111    dtype: string112  - name: corpus-id113    dtype: string114  - name: score115    dtype: int64116  splits:117  - name: test118    num_bytes: 92148119    num_examples: 3291120  download_size: 40493121  dataset_size: 92148122- config_name: javascript-queries123  features:124  - name: id125    dtype: string126  - name: text127    dtype: string128  splits:129  - name: test130    num_bytes: 1506447131    num_examples: 3291132  download_size: 637355133  dataset_size: 1506447134- config_name: php-corpus135  features:136  - name: id137    dtype: string138  - name: text139    dtype: string140  - name: title141    dtype: string142  splits:143  - name: test144    num_bytes: 70164589145    num_examples: 268237146  download_size: 27534526147  dataset_size: 70164589148- config_name: php-qrels149  features:150  - name: query-id151    dtype: string152  - name: corpus-id153    dtype: string154  - name: score155    dtype: int64156  splits:157  - name: test158    num_bytes: 420420159    num_examples: 14014160  download_size: 176431161  dataset_size: 420420162- config_name: php-queries163  features:164  - name: id165    dtype: string166  - name: text167    dtype: string168  splits:169  - name: test170    num_bytes: 4928215171    num_examples: 14014172  download_size: 1879778173  dataset_size: 4928215174- config_name: python-corpus175  features:176  - name: id177    dtype: string178  - name: text179    dtype: string180  - name: title181    dtype: string182  splits:183  - name: test184    num_bytes: 108454853185    num_examples: 280652186  download_size: 44304500187  dataset_size: 108454853188- config_name: python-qrels189  features:190  - name: query-id191    dtype: string192  - name: corpus-id193    dtype: string194  - name: score195    dtype: int64196  splits:197  - name: test198    num_bytes: 447540199    num_examples: 14918200  download_size: 185025201  dataset_size: 447540202- config_name: python-queries203  features:204  - name: id205    dtype: string206  - name: text207    dtype: string208  splits:209  - name: test210    num_bytes: 8455942211    num_examples: 14918212  download_size: 3542383213  dataset_size: 8455942214- config_name: ruby-corpus215  features:216  - name: id217    dtype: string218  - name: text219    dtype: string220  - name: title221    dtype: string222  splits:223  - name: test224    num_bytes: 5489759225    num_examples: 27588226  download_size: 2546854227  dataset_size: 5489759228- config_name: ruby-qrels229  features:230  - name: query-id231    dtype: string232  - name: corpus-id233    dtype: string234  - name: score235    dtype: int64236  splits:237  - name: test238    num_bytes: 35308239    num_examples: 1261240  download_size: 15588241  dataset_size: 35308242- config_name: ruby-queries243  features:244  - name: id245    dtype: string246  - name: text247    dtype: string248  splits:249  - name: test250    num_bytes: 354183251    num_examples: 1261252  download_size: 164078253  dataset_size: 354183254configs:255- config_name: go-corpus256  data_files:257  - split: test258    path: go-corpus/test-*259- config_name: go-qrels260  data_files:261  - split: test262    path: go-qrels/test-*263- config_name: go-queries264  data_files:265  - split: test266    path: go-queries/test-*267- config_name: java-corpus268  data_files:269  - split: test270    path: java-corpus/test-*271- config_name: java-qrels272  data_files:273  - split: test274    path: java-qrels/test-*275- config_name: java-queries276  data_files:277  - split: test278    path: java-queries/test-*279- config_name: javascript-corpus280  data_files:281  - split: test282    path: javascript-corpus/test-*283- config_name: javascript-qrels284  data_files:285  - split: test286    path: javascript-qrels/test-*287- config_name: javascript-queries288  data_files:289  - split: test290    path: javascript-queries/test-*291- config_name: php-corpus292  data_files:293  - split: test294    path: php-corpus/test-*295- config_name: php-qrels296  data_files:297  - split: test298    path: php-qrels/test-*299- config_name: php-queries300  data_files:301  - split: test302    path: php-queries/test-*303- config_name: python-corpus304  data_files:305  - split: test306    path: python-corpus/test-*307- config_name: python-qrels308  data_files:309  - split: test310    path: python-qrels/test-*311- config_name: python-queries312  data_files:313  - split: test314    path: python-queries/test-*315- config_name: ruby-corpus316  data_files:317  - split: test318    path: ruby-corpus/test-*319- config_name: ruby-qrels320  data_files:321  - split: test322    path: ruby-qrels/test-*323- config_name: ruby-queries324  data_files:325  - split: test326    path: ruby-queries/test-*327tags:328- mteb329- text330---331<!-- adapted from https://github.com/huggingface/huggingface_hub/blob/v0.30.2/src/huggingface_hub/templates/datasetcard_template.md -->332 333<div align="center" style="padding: 40px 20px; background-color: white; border-radius: 12px; box-shadow: 0 2px 10px rgba(0, 0, 0, 0.05); max-width: 600px; margin: 0 auto;">334  <h1 style="font-size: 3.5rem; color: #1a1a1a; margin: 0 0 20px 0; letter-spacing: 2px; font-weight: 700;">CodeSearchNetCCRetrieval</h1>335  <div style="font-size: 1.5rem; color: #4a4a4a; margin-bottom: 5px; font-weight: 300;">An <a href="https://github.com/embeddings-benchmark/mteb" style="color: #2c5282; font-weight: 600; text-decoration: none;" onmouseover="this.style.textDecoration='underline'" onmouseout="this.style.textDecoration='none'">MTEB</a> dataset</div>336  <div style="font-size: 0.9rem; color: #2c5282; margin-top: 10px;">Massive Text Embedding Benchmark</div>337</div>338 339The dataset is a collection of code snippets. The task is to retrieve the most relevant code snippet for a given code snippet.340 341|               |                                             |342|---------------|---------------------------------------------|343| Task category | t2t                              |344| Domains       | Programming, Written                               |345| Reference     | https://arxiv.org/abs/2407.02883 |346 347Source datasets:348- [CoIR-Retrieval/CodeSearchNet-ccr](https://huggingface.co/datasets/CoIR-Retrieval/CodeSearchNet-ccr)349 350 351## How to evaluate on this task352 353You can evaluate an embedding model on this dataset using the following code:354 355```python356import mteb357 358task = mteb.get_task("CodeSearchNetCCRetrieval")359evaluator = mteb.MTEB([task])360 361model = mteb.get_model(YOUR_MODEL)362evaluator.run(model)363```364 365<!-- Datasets want link to arxiv in readme to autolink dataset with paper -->366To learn more about how to run models on `mteb` task check out the [GitHub repository](https://github.com/embeddings-benchmark/mteb).367 368## Citation369 370If you use this dataset, please cite the dataset as well as [mteb](https://github.com/embeddings-benchmark/mteb), as this dataset likely includes additional processing as a part of the [MMTEB Contribution](https://github.com/embeddings-benchmark/mteb/tree/main/docs/mmteb).371 372```bibtex373 374@misc{li2024coircomprehensivebenchmarkcode,375  archiveprefix = {arXiv},376  author = {Xiangyang Li and Kuicai Dong and Yi Quan Lee and Wei Xia and Yichun Yin and Hao Zhang and Yong Liu and Yasheng Wang and Ruiming Tang},377  eprint = {2407.02883},378  primaryclass = {cs.IR},379  title = {CoIR: A Comprehensive Benchmark for Code Information Retrieval Models},380  url = {https://arxiv.org/abs/2407.02883},381  year = {2024},382}383 384 385@article{enevoldsen2025mmtebmassivemultilingualtext,386  title={MMTEB: Massive Multilingual Text Embedding Benchmark},387  author={Kenneth Enevoldsen and Isaac Chung and Imene Kerboua and Márton Kardos and Ashwin Mathur and David Stap and Jay Gala and Wissam Siblini and Dominik Krzemiński and Genta Indra Winata and Saba Sturua and Saiteja Utpala and Mathieu Ciancone and Marion Schaeffer and Gabriel Sequeira and Diganta Misra and Shreeya Dhakal and Jonathan Rystrøm and Roman Solomatin and Ömer Çağatan and Akash Kundu and Martin Bernstorff and Shitao Xiao and Akshita Sukhlecha and Bhavish Pahwa and Rafał Poświata and Kranthi Kiran GV and Shawon Ashraf and Daniel Auras and Björn Plüster and Jan Philipp Harries and Loïc Magne and Isabelle Mohr and Mariya Hendriksen and Dawei Zhu and Hippolyte Gisserot-Boukhlef and Tom Aarsen and Jan Kostkan and Konrad Wojtasik and Taemin Lee and Marek Šuppa and Crystina Zhang and Roberta Rocca and Mohammed Hamdy and Andrianos Michail and John Yang and Manuel Faysse and Aleksei Vatolin and Nandan Thakur and Manan Dey and Dipam Vasani and Pranjal Chitale and Simone Tedeschi and Nguyen Tai and Artem Snegirev and Michael Günther and Mengzhou Xia and Weijia Shi and Xing Han Lù and Jordan Clive and Gayatri Krishnakumar and Anna Maksimova and Silvan Wehrli and Maria Tikhonova and Henil Panchal and Aleksandr Abramov and Malte Ostendorff and Zheng Liu and Simon Clematide and Lester James Miranda and Alena Fenogenova and Guangyu Song and Ruqiya Bin Safi and Wen-Ding Li and Alessia Borghini and Federico Cassano and Hongjin Su and Jimmy Lin and Howard Yen and Lasse Hansen and Sara Hooker and Chenghao Xiao and Vaibhav Adlakha and Orion Weller and Siva Reddy and Niklas Muennighoff},388  publisher = {arXiv},389  journal={arXiv preprint arXiv:2502.13595},390  year={2025},391  url={https://arxiv.org/abs/2502.13595},392  doi = {10.48550/arXiv.2502.13595},393}394 395@article{muennighoff2022mteb,396  author = {Muennighoff, Niklas and Tazi, Nouamane and Magne, Loïc and Reimers, Nils},397  title = {MTEB: Massive Text Embedding Benchmark},398  publisher = {arXiv},399  journal={arXiv preprint arXiv:2210.07316},400  year = {2022}401  url = {https://arxiv.org/abs/2210.07316},402  doi = {10.48550/ARXIV.2210.07316},403}404```405 406# Dataset Statistics407<details>408  <summary> Dataset Statistics</summary>409 410The following code contains the descriptive statistics from the task. These can also be obtained using:411 412```python413import mteb414 415task = mteb.get_task("CodeSearchNetCCRetrieval")416 417desc_stats = task.metadata.descriptive_stats418```419 420```json421{422    "test": {423        "num_samples": 1058035,424        "number_of_characters": 284081702,425        "documents_text_statistics": {426            "total_text_length": 263684735,427            "min_text_length": 16,428            "average_text_length": 262.249182972409,429            "max_text_length": 139971,430            "unique_texts": 995481431        },432        "documents_image_statistics": null,433        "queries_text_statistics": {434            "total_text_length": 20396967,435            "min_text_length": 23,436            "average_text_length": 388.06276516809044,437            "max_text_length": 214210,438            "unique_texts": 52439439        },440        "queries_image_statistics": null,441        "relevant_docs_statistics": {442            "num_relevant_docs": 52561,443            "min_relevant_docs_per_query": 1,444            "average_relevant_docs_per_query": 1.0,445            "max_relevant_docs_per_query": 1,446            "unique_relevant_docs": 52561447        },448        "top_ranked_statistics": null,449        "hf_subset_descriptive_stats": {450            "python": {451                "num_samples": 295570,452                "number_of_characters": 109985967,453                "documents_text_statistics": {454                    "total_text_length": 101754313,455                    "min_text_length": 16,456                    "average_text_length": 362.56400453230333,457                    "max_text_length": 10111,458                    "unique_texts": 280036459                },460                "documents_image_statistics": null,461                "queries_text_statistics": {462                    "total_text_length": 8231654,463                    "min_text_length": 38,464                    "average_text_length": 551.7934039415471,465                    "max_text_length": 8326,466                    "unique_texts": 14918467                },468                "queries_image_statistics": null,469                "relevant_docs_statistics": {470                    "num_relevant_docs": 14918,471                    "min_relevant_docs_per_query": 1,472                    "average_relevant_docs_per_query": 1.0,473                    "max_relevant_docs_per_query": 1,474                    "unique_relevant_docs": 14918475                },476                "top_ranked_statistics": null477            },478            "javascript": {479                "num_samples": 68492,480                "number_of_characters": 18661880,481                "documents_text_statistics": {482                    "total_text_length": 17201640,483                    "min_text_length": 17,484                    "average_text_length": 263.8247879633748,485                    "max_text_length": 139971,486                    "unique_texts": 64779487                },488                "documents_image_statistics": null,489                "queries_text_statistics": {490                    "total_text_length": 1460240,491                    "min_text_length": 40,492                    "average_text_length": 443.70707991491946,493                    "max_text_length": 214210,494                    "unique_texts": 3291495                },496                "queries_image_statistics": null,497                "relevant_docs_statistics": {498                    "num_relevant_docs": 3291,499                    "min_relevant_docs_per_query": 1,500                    "average_relevant_docs_per_query": 1.0,501                    "max_relevant_docs_per_query": 1,502                    "unique_relevant_docs": 3291503                },504                "top_ranked_statistics": null505            },506            "go": {507                "num_samples": 190857,508                "number_of_characters": 33082384,509                "documents_text_statistics": {510                    "total_text_length": 31183720,511                    "min_text_length": 16,512                    "average_text_length": 170.64995758885817,513                    "max_text_length": 51245,514                    "unique_texts": 179845515                },516                "documents_image_statistics": null,517                "queries_text_statistics": {518                    "total_text_length": 1898664,519                    "min_text_length": 23,520                    "average_text_length": 233.76803742920464,521                    "max_text_length": 3589,522                    "unique_texts": 8122523                },524                "queries_image_statistics": null,525                "relevant_docs_statistics": {526                    "num_relevant_docs": 8122,527                    "min_relevant_docs_per_query": 1,528                    "average_relevant_docs_per_query": 1.0,529                    "max_relevant_docs_per_query": 1,530                    "unique_relevant_docs": 8122531                },532                "top_ranked_statistics": null533            },534            "ruby": {535                "num_samples": 28849,536                "number_of_characters": 5236077,537                "documents_text_statistics": {538                    "total_text_length": 4899550,539                    "min_text_length": 19,540                    "average_text_length": 177.59714368566043,541                    "max_text_length": 6201,542                    "unique_texts": 26997543                },544                "documents_image_statistics": null,545                "queries_text_statistics": {546                    "total_text_length": 336527,547                    "min_text_length": 36,548                    "average_text_length": 266.8731165741475,549                    "max_text_length": 2244,550                    "unique_texts": 1261551                },552                "queries_image_statistics": null,553                "relevant_docs_statistics": {554                    "num_relevant_docs": 1261,555                    "min_relevant_docs_per_query": 1,556                    "average_relevant_docs_per_query": 1.0,557                    "max_relevant_docs_per_query": 1,558                    "unique_relevant_docs": 1261559                },560                "top_ranked_statistics": null561            },562            "java": {563                "num_samples": 192016,564                "number_of_characters": 48626999,565                "documents_text_statistics": {566                    "total_text_length": 44874537,567                    "min_text_length": 21,568                    "average_text_length": 247.8420918916829,569                    "max_text_length": 15046,570                    "unique_texts": 178984571                },572                "documents_image_statistics": null,573                "queries_text_statistics": {574                    "total_text_length": 3752462,575                    "min_text_length": 38,576                    "average_text_length": 342.5341853035144,577                    "max_text_length": 5066,578                    "unique_texts": 10844579                },580                "queries_image_statistics": null,581                "relevant_docs_statistics": {582                    "num_relevant_docs": 10955,583                    "min_relevant_docs_per_query": 1,584                    "average_relevant_docs_per_query": 1.0,585                    "max_relevant_docs_per_query": 1,586                    "unique_relevant_docs": 10955587                },588                "top_ranked_statistics": null589            },590            "php": {591                "num_samples": 282251,592                "number_of_characters": 68488395,593                "documents_text_statistics": {594                    "total_text_length": 63770975,595                    "min_text_length": 20,596                    "average_text_length": 237.74115800579338,597                    "max_text_length": 6961,598                    "unique_texts": 264876599                },600                "documents_image_statistics": null,601                "queries_text_statistics": {602                    "total_text_length": 4717420,603                    "min_text_length": 40,604                    "average_text_length": 336.62194947909234,605                    "max_text_length": 2995,606                    "unique_texts": 14003607                },608                "queries_image_statistics": null,609                "relevant_docs_statistics": {610                    "num_relevant_docs": 14014,611                    "min_relevant_docs_per_query": 1,612                    "average_relevant_docs_per_query": 1.0,613                    "max_relevant_docs_per_query": 1,614                    "unique_relevant_docs": 14014615                },616                "top_ranked_statistics": null617            }618        }619    }620}621```622 623</details>624 625---626*This dataset card was automatically generated using [MTEB](https://github.com/embeddings-benchmark/mteb)*