Team Ai
Datasetpublic

AmazonScience/mxeval

A collection of execution-based multi-lingual benchmark for code generation.

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes52downloads
mxeval.py323 linesDownload Raw Back to root
1import json2import os3import requests4import datasets5 6import os7from collections import defaultdict8 9_CITATION = """\10@article{mbxp_athiwaratkun2022,11  title = {Multi-lingual Evaluation of Code Generation Models},12  author = {Athiwaratkun, Ben and13   Gouda, Sanjay Krishna and14   Wang, Zijian and15   Li, Xiaopeng and16   Tian, Yuchen and17   Tan, Ming18   and Ahmad, Wasi Uddin and19   Wang, Shiqi and20   Sun, Qing and21   Shang, Mingyue and22   Gonugondla, Sujan Kumar and23   Ding, Hantian and24   Kumar, Varun and25   Fulton, Nathan and26   Farahani, Arash and27   Jain, Siddhartha and28   Giaquinto, Robert and29   Qian, Haifeng and30   Ramanathan, Murali Krishna and31   Nallapati, Ramesh and32   Ray, Baishakhi and33   Bhatia, Parminder and34   Sengupta, Sudipta and35   Roth, Dan and36   Xiang, Bing},37  doi = {10.48550/ARXIV.2210.14868},38  url = {https://arxiv.org/abs/2210.14868},39  keywords = {Machine Learning (cs.LG), Computation and Language (cs.CL), FOS: Computer and information sciences, FOS: Computer and information sciences},40  publisher = {arXiv},41  year = {2022},42  copyright = {Creative Commons Attribution 4.0 International}43}"""44 45VERSION=f"1.1.0"46 47_HOMEPAGE = "https://github.com/amazon-science/mxeval"48 49_LICENSE = "Apache License 2.0"50 51_DESCRIPTION = """\52A collection of execution-based multi-lingual benchmark for code generation.53"""54 55_LICENSES = defaultdict(lambda: _LICENSE)56_LICENSES["humaneval_python"] = "MIT License"57_LICENSES["mbxp_python"] = "CC-BY-4.0"58 59_CITATIONS = defaultdict(lambda: _CITATION)60 61_CITATIONS["multi-humaneval"] = """\62@article{mbxp_athiwaratkun2022,63  title = {Multi-lingual Evaluation of Code Generation Models},64  author = {Athiwaratkun, Ben and65   Gouda, Sanjay Krishna and66   Wang, Zijian and67   Li, Xiaopeng and68   Tian, Yuchen and69   Tan, Ming70   and Ahmad, Wasi Uddin and71   Wang, Shiqi and72   Sun, Qing and73   Shang, Mingyue and74   Gonugondla, Sujan Kumar and75   Ding, Hantian and76   Kumar, Varun and77   Fulton, Nathan and78   Farahani, Arash and79   Jain, Siddhartha and80   Giaquinto, Robert and81   Qian, Haifeng and82   Ramanathan, Murali Krishna and83   Nallapati, Ramesh and84   Ray, Baishakhi and85   Bhatia, Parminder and86   Sengupta, Sudipta and87   Roth, Dan and88   Xiang, Bing},89  doi = {10.48550/ARXIV.2210.14868},90  url = {https://arxiv.org/abs/2210.14868},91  keywords = {Machine Learning (cs.LG), Computation and Language (cs.CL), FOS: Computer and information sciences, FOS: Computer and information sciences},92  publisher = {arXiv},93  year = {2022},94  copyright = {Creative Commons Attribution 4.0 International}95}96@misc{chen2021evaluating,97      title={Evaluating Large Language Models Trained on Code},98      author={Mark Chen and Jerry Tworek and Heewoo Jun and Qiming Yuan and Henrique Ponde de Oliveira Pinto and Jared Kaplan and Harri Edwards and Yuri Burda and Nicholas Joseph and Greg Brockman and Alex Ray and Raul Puri and Gretchen Krueger and Michael Petrov and Heidy Khlaaf and Girish Sastry and Pamela Mishkin and Brooke Chan and Scott Gray and Nick Ryder and Mikhail Pavlov and Alethea Power and Lukasz Kaiser and Mohammad Bavarian and Clemens Winter and Philippe Tillet and Felipe Petroski Such and Dave Cummings and Matthias Plappert and Fotios Chantzis and Elizabeth Barnes and Ariel Herbert-Voss and William Hebgen Guss and Alex Nichol and Alex Paino and Nikolas Tezak and Jie Tang and Igor Babuschkin and Suchir Balaji and Shantanu Jain and William Saunders and Christopher Hesse and Andrew N. Carr and Jan Leike and Josh Achiam and Vedant Misra and Evan Morikawa and Alec Radford and Matthew Knight and Miles Brundage and Mira Murati and Katie Mayer and Peter Welinder and Bob McGrew and Dario Amodei and Sam McCandlish and Ilya Sutskever and Wojciech Zaremba},99      year={2021},100      eprint={2107.03374},101      archivePrefix={arXiv},102      primaryClass={cs.LG}103}"""104 105_CITATIONS["mbxp"] = """\106@article{mbxp_athiwaratkun2022,107  title = {Multi-lingual Evaluation of Code Generation Models},108  author = {Athiwaratkun, Ben and109   Gouda, Sanjay Krishna and110   Wang, Zijian and111   Li, Xiaopeng and112   Tian, Yuchen and113   Tan, Ming114   and Ahmad, Wasi Uddin and115   Wang, Shiqi and116   Sun, Qing and117   Shang, Mingyue and118   Gonugondla, Sujan Kumar and119   Ding, Hantian and120   Kumar, Varun and121   Fulton, Nathan and122   Farahani, Arash and123   Jain, Siddhartha and124   Giaquinto, Robert and125   Qian, Haifeng and126   Ramanathan, Murali Krishna and127   Nallapati, Ramesh and128   Ray, Baishakhi and129   Bhatia, Parminder and130   Sengupta, Sudipta and131   Roth, Dan and132   Xiang, Bing},133  doi = {10.48550/ARXIV.2210.14868},134  url = {https://arxiv.org/abs/2210.14868},135  keywords = {Machine Learning (cs.LG), Computation and Language (cs.CL), FOS: Computer and information sciences, FOS: Computer and information sciences},136  publisher = {arXiv},137  year = {2022},138  copyright = {Creative Commons Attribution 4.0 International}139}140@article{austin2021program,141  title={Program Synthesis with Large Language Models},142  author={Austin, Jacob and Odena, Augustus and Nye, Maxwell and Bosma, Maarten and Michalewski, Henryk and Dohan, David and Jiang, Ellen and Cai, Carrie and Terry, Michael and Le, Quoc and others},143  journal={arXiv preprint arXiv:2108.07732},144  year={2021}145}"""146 147_CITATIONS["mathqa-x"] = """\148@article{mbxp_athiwaratkun2022,149  title = {Multi-lingual Evaluation of Code Generation Models},150  author = {Athiwaratkun, Ben and151   Gouda, Sanjay Krishna and152   Wang, Zijian and153   Li, Xiaopeng and154   Tian, Yuchen and155   Tan, Ming156   and Ahmad, Wasi Uddin and157   Wang, Shiqi and158   Sun, Qing and159   Shang, Mingyue and160   Gonugondla, Sujan Kumar and161   Ding, Hantian and162   Kumar, Varun and163   Fulton, Nathan and164   Farahani, Arash and165   Jain, Siddhartha and166   Giaquinto, Robert and167   Qian, Haifeng and168   Ramanathan, Murali Krishna and169   Nallapati, Ramesh and170   Ray, Baishakhi and171   Bhatia, Parminder and172   Sengupta, Sudipta and173   Roth, Dan and174   Xiang, Bing},175  doi = {10.48550/ARXIV.2210.14868},176  url = {https://arxiv.org/abs/2210.14868},177  keywords = {Machine Learning (cs.LG), Computation and Language (cs.CL), FOS: Computer and information sciences, FOS: Computer and information sciences},178  publisher = {arXiv},179  year = {2022},180  copyright = {Creative Commons Attribution 4.0 International}181}182@inproceedings{amini-etal-2019-mathqa,183    title={MathQA: Towards Interpretable Math Word Problem Solving with Operation-Based Formalisms},184    author={Amini, Aida  and185      Gabriel, Saadia  and186      Lin, Shanchuan  and187      Koncel-Kedziorski, Rik  and188      Choi, Yejin  and189      Hajishirzi, Hannaneh},190    booktitle={Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers)},191    month={jun},192    year= {2019},193    address = {Minneapolis, Minnesota},194    publisher = {Association for Computational Linguistics},195    url={https://aclanthology.org/N19-1245}196    doi={10.18653/v1/N19-1245},197    pages={2357--2367},198}199"""200 201_DATASET_NAME_MAPPER = {202    "mbxp": "mbxp",203    "multi-humaneval": "multilingual_humaneval",204    "mathqa-x": "multilingual_mathqa"205}206 207_GITHUB_ROOT = "https://raw.githubusercontent.com/amazon-science/mxeval/main/data/"208 209 210def get_metadata_dict(dataset):211    metadata_dict_path = requests.get(os.path.join(_GITHUB_ROOT, dataset, "metadata.json"))212    metadata = json.loads(metadata_dict_path.text)213    return metadata214 215 216MBXP_LANGUAGES = get_metadata_dict("mbxp")217 218MATHQA_LANGUAGES = get_metadata_dict("multilingual_mathqa")219 220HUMANEVAL_LANGUAGES = get_metadata_dict("multilingual_humaneval")221 222_DATASET_LANGS = {223    "multi-humaneval": HUMANEVAL_LANGUAGES,224    "mathqa-x": MATHQA_LANGUAGES,225    "mbxp": MBXP_LANGUAGES226}227 228_INTERNAL_DATASET_NAMES = {229    "multi-humaneval": "multilingual_humaneval",230    "mathqa-x": "multilingual_mathqa",231    "mbxp": "mbxp"232}233 234_URL_DICT = {235    f"{dataset.lower()}_{language.lower()}": os.path.join(236        _GITHUB_ROOT,237        _INTERNAL_DATASET_NAMES[dataset],238        _DATASET_LANGS[dataset][language]239        )240    for dataset, languages in _DATASET_LANGS.items() for language in languages241}242 243 244class MxEvalConfig(datasets.BuilderConfig):245    """BuilderConfig for MxEval."""246 247    def __init__(248        self,249        dataset,250        citation,251        version,252        **kwargs,253    ):254        super(MxEvalConfig, self).__init__(version=datasets.Version(f"{version}", ""), **kwargs)255        self.dataset_name = dataset256        self.data_dir = os.path.join(_GITHUB_ROOT, dataset)257        self.citation = citation258 259 260class MxEval(datasets.GeneratorBasedBuilder):261    """MxEval: An execution-based multiLingual benchmark for code generation."""262 263    BUILDER_CONFIGS = [264        MxEvalConfig(265            name=f"{dataset}",266            version=VERSION,267            citation=_CITATIONS[f"{dataset}"],268            dataset=_DATASET_NAME_MAPPER[dataset],269            description=f"Benchmark for {dataset}",270        ) for dataset in _DATASET_LANGS271    ]272 273    def _info(self):274        self.build_name = self.name275        features = datasets.Features(276            {277                "task_id": datasets.Value("string"),278                "language": datasets.Value("string"),279                "prompt": datasets.Value("string"),280                "description": datasets.Value("string"),281                "test": datasets.Value("string"),282                "entry_point": datasets.Value("string"),283                "canonical_solution": datasets.Value("string"),284            }285        )286 287        return datasets.DatasetInfo(288            description=_DESCRIPTION,289            features=features,290            supervised_keys=None,291            homepage=_HOMEPAGE,292            license=_LICENSES[self.config.name],293            citation=_CITATIONS[self.config.name],294        )295 296 297    def _split_generators(298            self, dl_manager299    ):300        """Returns SplitGenerators."""301        return [302            datasets.SplitGenerator(303                name=datasets.Split(lang),304                gen_kwargs={305                    "filepath": dl_manager.download_and_extract(306                        url_or_urls=_URL_DICT[f"{self.config.name}_{lang}"]307                    ),308                },309            ) for lang in _DATASET_LANGS[self.config.name]310        ]311 312    313    def _generate_examples(self, filepath):314        """Yields examples."""315        with open(filepath) as file:316            data = []317            for line in file:318                jd = json.loads(line)319                data.append(jd)320            id_ = 0321            for sample in data:322                yield id_, sample323                id_ += 1