Team Ai
Datasetpublic

CodedotAI/code_clippy_github

The Code Clippy dataset consists of various public codebases from GitHub in 22 programming languages with 23 extensions totalling about 16 TB of data when uncompressed. The dataset was created from the public GitHub dataset on Google BiqQuery.

sourceHugging Facemitupdated 4y agoView on Hugging Face
20likes3.4kdownloads
code_clippy_github.py208 linesDownload Raw Back to root
1# coding=utf-82# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8#     http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15 16# Special thanks to @lvwerra -- we reference his repository here: https://huggingface.co/datasets/lvwerra/github-code/17"""Code Clippy Github Code dataset."""18 19import os20 21 22import datasets23from datasets.download.streaming_download_manager import xopen24from huggingface_hub import HfApi, HfFolder25from datasets.data_files import DataFilesDict26 27import gzip28import json29 30_REPO_NAME = "CodedotAI/code_clippy_github"31 32_LANG_TO_EXTENSION = {33    "C": [".c"],34    "C#": [".cs"],35    "C++": [".cpp"],36    "CSS": [".css"],37    "Dart" : [".dart"],38    "GO": [".go"],39    "HTML":[".html"],40    "Java": [".java"],41    "JavaScript": [".js"],42    "Jupyter Notebooks (Python)": [".ipynb"],43    "Kotlin" : [".kt"],44    "Lisp" : [".lisp"],45    "Matlab" : [".m"],46    "PHP": [".php"],47    "Perl": [".pl"],48    "Python": [".py"],49    "R" : [".r"],50    "Ruby": [".rb"],51    "Rust": [".rs"],52    "SQL": [".sql"],53    "Shell": [".sh"],54    "Swift" : [".swift"],55    "TypeScript": [".ts"],56}57 58_LICENSES = [59    'mit',60    'apache-2.0',61    'gpl-2.0',62    'gpl-3.0',63    'bsd-3-clause',64    'bsd-2-clause',65    'unlicense',66    'apacheagpl-3.0',67    'lgpl-3.0',68    'cc0-1.0',69    'epl-1.0',70    'lgpl-2.1',71    'mpl-2.0',72    'isc',73    'artistic-2.0'74 ]75 76_DESCRIPTION = """\77The Code Clippy dataset consists of various public codebases from GitHub in 22 programming languages with 23 extensions \78    totalling about 16 TB of data when uncompressed. The dataset was created from the public GitHub dataset on Google BiqQuery.79"""80 81_HOMEPAGE = "https://cloud.google.com/blog/topics/public-datasets/github-on-bigquery-analyze-all-the-open-source-code/"82 83 84_EXTENSION_TO_LANG = {}85for lang in _LANG_TO_EXTENSION:86    for extension in _LANG_TO_EXTENSION[lang]:87        _EXTENSION_TO_LANG[extension] = lang88 89 90        91_LANG_CONFIGS = ["all"] + list(_LANG_TO_EXTENSION.keys())92_LICENSE_CONFIGS = ["all"] + _LICENSES93        94class CodeClippyGithubConfig(datasets.BuilderConfig):95    """BuilderConfig for the Code Clippy Github dataset."""96 97    def __init__(self, *args, languages=["all"], licenses=["all"], **kwargs):98        """BuilderConfig for the Code Clippy Github dataset.99        Args:100            languages (:obj:`List[str]`): List of languages to load.101            licenses (:obj:`List[str]`): List of licenses to load.102            **kwargs: keyword arguments forwarded to super.103        """104        super().__init__(105            *args,106            name="+".join(languages)+"-"+"+".join(licenses),107            **kwargs,108        )109        110        languages = set(languages)111        licenses = set(licenses)112        113        assert all([language in _LANG_CONFIGS for language in languages]), f"Language not in {_LANG_CONFIGS}."114        assert all([license in _LICENSE_CONFIGS for license in licenses]), f"License not in {_LICENSE_CONFIGS}."115        116        if "all" in languages:117            assert len(languages)==1, "Passed 'all' together with other languages."118            self.filter_languages = False119        else:120            self.filter_languages = True121            122        if "all" in licenses:123            assert len(licenses)==1, "Passed 'all' together with other licenses."124            self.filter_licenses = False125        else:126            self.filter_licenses = True127        128        self.languages = set(languages)129        self.licenses = set(licenses)130 131 132        133class CodeClippyGithub(datasets.GeneratorBasedBuilder):134    """Code Clippy Github dataset."""135 136    VERSION = datasets.Version("1.0.0")137    138    BUILDER_CONFIG_CLASS = CodeClippyGithubConfig139    BUILDER_CONFIGS = [CodeClippyGithubConfig(languages=[lang], licenses=[license]) for lang in _LANG_CONFIGS140                                                                        for license in _LICENSE_CONFIGS]141    DEFAULT_CONFIG_NAME = "all-all"142    143    144    def _info(self):145        return datasets.DatasetInfo(146            description=_DESCRIPTION,147            features=datasets.Features({"code_text": datasets.Value("string"),148                                        "repo_name": datasets.Value("string"),149                                        "file_path": datasets.Value("string"), 150                                        "language": datasets.Value("string"),151                                        "license": datasets.Value("string"),152                                        "size": datasets.Value("int32")}),153            supervised_keys=None,154            homepage=_HOMEPAGE,155            license="Multiple: see the 'license' field of each sample.",156            157        )158 159    def _split_generators(self, dl_manager):160        num_shards = 50_000161        data_files = [162            f"github-dedup-{_index:012d}.json.gz"163            for _index in range(num_shards)164        ]165        files = dl_manager.download(data_files)166        return [167            datasets.SplitGenerator(168                name=datasets.Split.TRAIN,169                gen_kwargs={170                    "files": files,171                },172            ),173        ]174 175    def _generate_examples(self, files):176        key = 0177        for file_idx, file in enumerate(files):178            with xopen(file, "rb") as file:  # download file if in streaming mode179                with gzip.open(file, "rb") as f:180 181                    uncompressed_data = f.readlines()182 183                    for batch_idx, code_base in enumerate(uncompressed_data):184                        j_dict = json.loads(code_base.decode('utf-8'))185 186 187 188                        lang = lang_from_name(j_dict['path'])189                        license = j_dict["license"]190 191                        if self.config.filter_languages and not lang in self.config.languages:192                            continue193                        if self.config.filter_licenses and not license in self.config.licenses:194                            continue195                        # TODO: Add more features like header comments, filename, and other features useful in a prompt.196                        yield key, {"code_text": j_dict['content'],197                                    "repo_name": j_dict['repo_name'],198                                    "file_path": j_dict['path'],199                                    "license": license,200                                    "language": lang,201                                    "size": int(j_dict['f0_'])}202                        key += 1203 204                        205def lang_from_name(name):206    for extension in _EXTENSION_TO_LANG:207        if name.endswith(extension):208            return _EXTENSION_TO_LANG[extension]