CodedotAI/code_clippy_github
The Code Clippy dataset consists of various public codebases from GitHub in 22 programming languages with 23 extensions totalling about 16 TB of data when uncompressed. The dataset was created from the public GitHub dataset on Google BiqQuery.
203.4k
1# coding=utf-82# Copyright 2020 The HuggingFace Datasets Authors and the current dataset script contributor.3#4# Licensed under the Apache License, Version 2.0 (the "License");5# you may not use this file except in compliance with the License.6# You may obtain a copy of the License at7#8# http://www.apache.org/licenses/LICENSE-2.09#10# Unless required by applicable law or agreed to in writing, software11# distributed under the License is distributed on an "AS IS" BASIS,12# WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.13# See the License for the specific language governing permissions and14# limitations under the License.15 16# Special thanks to @lvwerra -- we reference his repository here: https://huggingface.co/datasets/lvwerra/github-code/17"""Code Clippy Github Code dataset."""18 19import os20 21 22import datasets23from datasets.download.streaming_download_manager import xopen24from huggingface_hub import HfApi, HfFolder25from datasets.data_files import DataFilesDict26 27import gzip28import json29 30_REPO_NAME = "CodedotAI/code_clippy_github"31 32_LANG_TO_EXTENSION = {33 "C": [".c"],34 "C#": [".cs"],35 "C++": [".cpp"],36 "CSS": [".css"],37 "Dart" : [".dart"],38 "GO": [".go"],39 "HTML":[".html"],40 "Java": [".java"],41 "JavaScript": [".js"],42 "Jupyter Notebooks (Python)": [".ipynb"],43 "Kotlin" : [".kt"],44 "Lisp" : [".lisp"],45 "Matlab" : [".m"],46 "PHP": [".php"],47 "Perl": [".pl"],48 "Python": [".py"],49 "R" : [".r"],50 "Ruby": [".rb"],51 "Rust": [".rs"],52 "SQL": [".sql"],53 "Shell": [".sh"],54 "Swift" : [".swift"],55 "TypeScript": [".ts"],56}57 58_LICENSES = [59 'mit',60 'apache-2.0',61 'gpl-2.0',62 'gpl-3.0',63 'bsd-3-clause',64 'bsd-2-clause',65 'unlicense',66 'apacheagpl-3.0',67 'lgpl-3.0',68 'cc0-1.0',69 'epl-1.0',70 'lgpl-2.1',71 'mpl-2.0',72 'isc',73 'artistic-2.0'74 ]75 76_DESCRIPTION = """\77The Code Clippy dataset consists of various public codebases from GitHub in 22 programming languages with 23 extensions \78 totalling about 16 TB of data when uncompressed. The dataset was created from the public GitHub dataset on Google BiqQuery.79"""80 81_HOMEPAGE = "https://cloud.google.com/blog/topics/public-datasets/github-on-bigquery-analyze-all-the-open-source-code/"82 83 84_EXTENSION_TO_LANG = {}85for lang in _LANG_TO_EXTENSION:86 for extension in _LANG_TO_EXTENSION[lang]:87 _EXTENSION_TO_LANG[extension] = lang88 89 90 91_LANG_CONFIGS = ["all"] + list(_LANG_TO_EXTENSION.keys())92_LICENSE_CONFIGS = ["all"] + _LICENSES93 94class CodeClippyGithubConfig(datasets.BuilderConfig):95 """BuilderConfig for the Code Clippy Github dataset."""96 97 def __init__(self, *args, languages=["all"], licenses=["all"], **kwargs):98 """BuilderConfig for the Code Clippy Github dataset.99 Args:100 languages (:obj:`List[str]`): List of languages to load.101 licenses (:obj:`List[str]`): List of licenses to load.102 **kwargs: keyword arguments forwarded to super.103 """104 super().__init__(105 *args,106 name="+".join(languages)+"-"+"+".join(licenses),107 **kwargs,108 )109 110 languages = set(languages)111 licenses = set(licenses)112 113 assert all([language in _LANG_CONFIGS for language in languages]), f"Language not in {_LANG_CONFIGS}."114 assert all([license in _LICENSE_CONFIGS for license in licenses]), f"License not in {_LICENSE_CONFIGS}."115 116 if "all" in languages:117 assert len(languages)==1, "Passed 'all' together with other languages."118 self.filter_languages = False119 else:120 self.filter_languages = True121 122 if "all" in licenses:123 assert len(licenses)==1, "Passed 'all' together with other licenses."124 self.filter_licenses = False125 else:126 self.filter_licenses = True127 128 self.languages = set(languages)129 self.licenses = set(licenses)130 131 132 133class CodeClippyGithub(datasets.GeneratorBasedBuilder):134 """Code Clippy Github dataset."""135 136 VERSION = datasets.Version("1.0.0")137 138 BUILDER_CONFIG_CLASS = CodeClippyGithubConfig139 BUILDER_CONFIGS = [CodeClippyGithubConfig(languages=[lang], licenses=[license]) for lang in _LANG_CONFIGS140 for license in _LICENSE_CONFIGS]141 DEFAULT_CONFIG_NAME = "all-all"142 143 144 def _info(self):145 return datasets.DatasetInfo(146 description=_DESCRIPTION,147 features=datasets.Features({"code_text": datasets.Value("string"),148 "repo_name": datasets.Value("string"),149 "file_path": datasets.Value("string"), 150 "language": datasets.Value("string"),151 "license": datasets.Value("string"),152 "size": datasets.Value("int32")}),153 supervised_keys=None,154 homepage=_HOMEPAGE,155 license="Multiple: see the 'license' field of each sample.",156 157 )158 159 def _split_generators(self, dl_manager):160 num_shards = 50_000161 data_files = [162 f"github-dedup-{_index:012d}.json.gz"163 for _index in range(num_shards)164 ]165 files = dl_manager.download(data_files)166 return [167 datasets.SplitGenerator(168 name=datasets.Split.TRAIN,169 gen_kwargs={170 "files": files,171 },172 ),173 ]174 175 def _generate_examples(self, files):176 key = 0177 for file_idx, file in enumerate(files):178 with xopen(file, "rb") as file: # download file if in streaming mode179 with gzip.open(file, "rb") as f:180 181 uncompressed_data = f.readlines()182 183 for batch_idx, code_base in enumerate(uncompressed_data):184 j_dict = json.loads(code_base.decode('utf-8'))185 186 187 188 lang = lang_from_name(j_dict['path'])189 license = j_dict["license"]190 191 if self.config.filter_languages and not lang in self.config.languages:192 continue193 if self.config.filter_licenses and not license in self.config.licenses:194 continue195 # TODO: Add more features like header comments, filename, and other features useful in a prompt.196 yield key, {"code_text": j_dict['content'],197 "repo_name": j_dict['repo_name'],198 "file_path": j_dict['path'],199 "license": license,200 "language": lang,201 "size": int(j_dict['f0_'])}202 key += 1203 204 205def lang_from_name(name):206 for extension in _EXTENSION_TO_LANG:207 if name.endswith(extension):208 return _EXTENSION_TO_LANG[extension]