Team Ai
Datasetpublic

facebook/neural_code_search

Neural-Code-Search-Evaluation-Dataset presents an evaluation dataset consisting of natural language query and code snippet pairs and a search corpus consisting of code snippets collected from the most popular Android repositories on GitHub.

sourceHugging Facecc-by-nc-4.0updated 3y agoView on Hugging Face
12likes392downloads
neural_code_search.py175 linesDownload Raw Back to root
1# coding=utf-82# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8#     http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15"""Neural-Code-Search-Evaluation-Dataset presents an evaluation dataset consisting of natural language query and code snippet pairs"""16 17 18import json19from itertools import chain20 21import datasets22 23 24_CITATION = """\25@InProceedings{huggingface:dataset,26title         = {Neural Code Search Evaluation Dataset},27authors       = {Hongyu Li, Seohyun Kim and Satish Chandra},28journal       = {arXiv e-prints},29year          = 2018,30eid           = {arXiv:1908.09804 [cs.SE]},31pages         = {arXiv:1908.09804 [cs.SE]},32archivePrefix = {arXiv},33eprint        = {1908.09804},34}35"""36 37_DESCRIPTION = """\38Neural-Code-Search-Evaluation-Dataset presents an evaluation dataset \39consisting of natural language query and code snippet pairs and a search corpus \40consisting of code snippets collected from the most popular Android repositories \41on GitHub.42"""43 44_HOMEPAGE = "https://github.com/facebookresearch/Neural-Code-Search-Evaluation-Dataset/tree/master/data"45 46_LICENSE = "CC-BY-NC 4.0 (Attr Non-Commercial Inter.)"47 48_BASE_URL = "https://raw.githubusercontent.com/facebookresearch/Neural-Code-Search-Evaluation-Dataset/master/data/"49_URLs = {50    "evaluation_dataset": _BASE_URL + "287_android_questions.json",51    "search_corpus_1": _BASE_URL + "search_corpus_1.tar.gz",52    "search_corpus_2": _BASE_URL + "search_corpus_2.tar.gz",53}54 55 56class NeuralCodeSearch(datasets.GeneratorBasedBuilder):57    """Neural Code Search Evaluation Dataset"""58 59    VERSION = datasets.Version("1.1.0")60 61    BUILDER_CONFIGS = [62        datasets.BuilderConfig(63            name="evaluation_dataset",64            version=VERSION,65            description="The evaluation dataset is composed of \66            287 Stack Overflow question and answer pairs",67        ),68        datasets.BuilderConfig(69            name="search_corpus",70            version=VERSION,71            description="The search corpus is indexed using all \72            method bodies parsed from the 24,549 GitHub repositories.",73        ),74    ]75 76    FILENAME_MAP = {77        "evaluation_dataset": "287_android_questions.json",78        "search_corpus": "search_corpus_1.jsonl",79    }80 81    def _info(self):82        if self.config.name == "evaluation_dataset":83            features = datasets.Features(84                {85                    "stackoverflow_id": datasets.Value("int32"),86                    "question": datasets.Value("string"),87                    "question_url": datasets.Value("string"),88                    "question_author": datasets.Value("string"),89                    "question_author_url": datasets.Value("string"),90                    "answer": datasets.Value("string"),91                    "answer_url": datasets.Value("string"),92                    "answer_author": datasets.Value("string"),93                    "answer_author_url": datasets.Value("string"),94                    "examples": datasets.features.Sequence(datasets.Value("int32")),95                    "examples_url": datasets.features.Sequence(datasets.Value("string")),96                }97            )98        else:99            features = datasets.Features(100                {101                    "id": datasets.Value("int32"),102                    "filepath": datasets.Value("string"),103                    "method_name": datasets.Value("string"),104                    "start_line": datasets.Value("int32"),105                    "end_line": datasets.Value("int32"),106                    "url": datasets.Value("string"),107                }108            )109 110        return datasets.DatasetInfo(111            description=_DESCRIPTION,112            features=features,113            supervised_keys=None,114            homepage=_HOMEPAGE,115            license=_LICENSE,116            citation=_CITATION,117        )118 119    def _split_generators(self, dl_manager):120        """Returns SplitGenerators."""121        if self.config.name == "evaluation_dataset":122            filepath = dl_manager.download_and_extract(_URLs[self.config.name])123            return [124                datasets.SplitGenerator(125                    name=datasets.Split.TRAIN,126                    gen_kwargs={"filepath": filepath},127                ),128            ]129        else:130            my_urls = [url for config, url in _URLs.items() if config.startswith(self.config.name)]131            archives = dl_manager.download(my_urls)132            return [133                datasets.SplitGenerator(134                    name=datasets.Split.TRAIN,135                    gen_kwargs={136                        "files": chain(*(dl_manager.iter_archive(archive) for archive in archives)),137                    },138                ),139            ]140 141    def _generate_examples(self, filepath=None, files=None):142        """Yields examples."""143        id_ = 0144        if self.config.name == "evaluation_dataset":145            with open(filepath, encoding="utf-8") as f:146                data = json.load(f)147                for row in data:148                    yield id_, {149                        "stackoverflow_id": row["stackoverflow_id"],150                        "question": row["question"],151                        "question_url": row["question_url"],152                        "question_author": row["question_author"],153                        "question_author_url": row["question_author_url"],154                        "answer": row["answer"],155                        "answer_url": row["answer_url"],156                        "answer_author": row["answer_author"],157                        "answer_author_url": row["answer_author_url"],158                        "examples": row["examples"],159                        "examples_url": row["examples_url"],160                    }161                    id_ += 1162        else:163            for _, f in files:164                for row in f:165                    data_dict = json.loads(row.decode("utf-8"))166                    yield id_, {167                        "id": data_dict["id"],168                        "filepath": data_dict["filepath"],169                        "method_name": data_dict["method_name"],170                        "start_line": data_dict["start_line"],171                        "end_line": data_dict["end_line"],172                        "url": data_dict["url"],173                    }174                    id_ += 1175