bigcode/commitpack
CommitPack is is a 4TB dataset of commits scraped from GitHub repositories that are permissively licensed.
7915k
1"""CommitPack"""2 3import json4import datasets5 6 7logger = datasets.logging.get_logger(__name__)8 9### To create paths ###10def get_paths():11 import json, glob, os12 files = {}13 for lang_dir in os.listdir("./data"):14 print("Processing", lang_dir)15 if not os.path.isdir("data/" + lang_dir):16 print(f"Skipping {lang_dir} as it is not a directory")17 continue18 for file in glob.glob(f"data/{lang_dir}/*.jsonl"):19 files[lang_dir] = files.get(lang_dir, []) + [file]20 with open(f"paths.json", "w") as f:21 json.dump(files, f)22 return files23 24_CITATION = """\25@article{muennighoff2023octopack,26 title={OctoPack: Instruction Tuning Code Large Language Models}, 27 author={Niklas Muennighoff and Qian Liu and Armel Zebaze and Qinkai Zheng and Binyuan Hui and Terry Yue Zhuo and Swayam Singh and Xiangru Tang and Leandro von Werra and Shayne Longpre},28 journal={arXiv preprint arXiv:2308.07124},29 year={2023}30}31"""32 33_DESCRIPTION = """\34CommitPack is is a 4TB dataset of commits scraped from GitHub repositories that are permissively licensed.35"""36 37URL = "https://huggingface.co/datasets/bigcode/commitpack/resolve/main/paths.json"38 39_LANG = ["json", "xml", "text", "javascript", "objective-c++", "python", "c", "c++", "markdown", "java", "html", "yaml", "go", "csv", "php", "jupyter-notebook", "gettext-catalog", "sql", "unity3d-asset", "typescript", "web-ontology-language", "ruby", "c#", "nix", "shell", "perl", "tex", "css", "restructuredtext", "rust", "groff", "ini", "scala", "coffeescript", "haskell", "swift", "lua", "svg", "gas", "ocaml", "erlang", "makefile", "asciidoc", "emacs-lisp", "scss", "clojure", "org", "common-lisp", "diff", "groovy", "html+erb", "nesc", "dart", "powershell", "f#", "dm", "kotlin", "pascal", "jsx", "viml", "actionscript", "cython", "turtle", "less", "mathematica", "xslt", "scheme", "perl6", "edn", "fortran", "java-server-pages", "standard-ml", "cmake", "json5", "vala", "vue", "freemarker", "graphql", "twig", "tcl", "pod", "dockerfile", "yacc", "postscript", "racket", "eagle", "haxe", "julia", "handlebars", "smarty", "visual-basic", "literate-haskell", "smalltalk", "isabelle", "nimrod", "zig", "m4", "max", "elixir", "mako", "arduino", "jade", "haml", "elm", "purebasic", "coldfusion", "lean", "r", "cuda", "textile", "robotframework", "abap", "rdoc", "llvm", "ada", "batchfile", "qml", "jasmin", "assembly", "g-code", "cucumber", "html+php", "kicad", "api-blueprint", "eiffel", "toml", "modelica", "bitbake", "lex", "stylus", "protocol-buffer", "unknown", "nit", "factor", "xs", "sass", "parrot-internal-representation", "html+django", "mediawiki", "logos", "genshi", "coldfusion-cfc", "xtend", "sqf", "vhdl", "antlr", "systemverilog", "hcl", "asp", "nsis", "inform-7", "slim", "groovy-server-pages", "ceylon", "fish", "processing", "component-pascal", "lasso", "glsl", "saltstack", "xbase", "autohotkey", "liquid", "purescript", "agda", "inno-setup", "oz", "chapel", "arc", "opencl", "graphviz-dot", "pawn", "jsoniq", "bluespec", "smali", "krl", "maple", "unrealscript", "ooc", "pure-data", "xquery", "digital-command-language", "moonscript", "awk", "pike", "livescript", "solidity", "monkey", "jsonld", "zephir", "crystal", "rhtml", "stata", "idris", "raml", "openscad", "red", "c2hs-haskell", "cycript", "applescript", "mupad", "literate-agda", "boo", "sourcepawn", "qmake", "ragel-in-ruby-host", "io", "desktop", "propeller-spin", "thrift", "volt", "xproc", "igor-pro", "lolcode", "html+eex", "logtalk", "mirah", "gnuplot", "literate-coffeescript", "jflex", "emberscript", "cobol", "yang", "rebol", "linker-script", "cartocss", "urweb", "rmarkdown", "darcs-patch", "csound", "squirrel", "apl", "hlsl", "latte", "pony", "ioke", "hy", "uno", "pan", "xojo", "papyrus", "stan", "slash", "supercollider", "vcl", "smt", "glyph", "wisp", "renpy", "clips", "dns-zone", "sas", "rouge", "ec", "dylan", "tcsh", "aspectj", "netlogo", "gap", "fancy", "coq", "click", "capn-proto", "flux", "forth", "ats", "netlinx", "clean", "parrot-assembly", "alloy", "lfe", "gdscript", "augeas", "sparql", "lilypond", "scilab", "autoit", "myghty", "blitzmax", "creole", "harbour", "piglatin", "opa", "sage", "ston", "maxscript", "lsl", "gentoo-ebuild", "nu", "bro", "xc", "j", "metal", "module-management-system", "webidl", "tea", "redcode", "shen", "pov-ray-sdl", "x10", "brainfuck", "ninja", "golo", "webassembly", "self", "labview", "octave", "pogoscript", "d", "http", "ecl", "chuck", "gosu", "parrot", "opal", "objective-j", "kit", "gams", "prolog", "clarion", "mask", "brightscript", "scaml", "matlab", "idl", "ags-script", "lookml", "apacheconf", "oxygene", "txl", "grammatical-framework", "renderscript", "mtml", "unified-parallel-c", "dogescript", "gentoo-eclass", "zimpl", "irc-log", "fantom", "numpy", "cirru", "xpages", "nginx", "objdump", "python-traceback", "realbasic", "befunge", "bison", "m", "omgrofl"]40_LICENSE = "Apache License 2.0"41_VERSION = datasets.Version("1.0.0", "")42 43 44class CommitPack(datasets.GeneratorBasedBuilder):45 BUILDER_CONFIGS = [46 datasets.BuilderConfig(47 name=lang,48 description=f"CommitPack {lang}",49 version=_VERSION,50 )51 for lang in _LANG52 ]53 54 def _info(self):55 return datasets.DatasetInfo(56 description=_DESCRIPTION,57 features=datasets.Features(58 {59 "commit": datasets.Value("string"),60 "old_file": datasets.Value("string"),61 "new_file": datasets.Value("string"),62 "old_contents": datasets.Value("string"),63 "new_contents": datasets.Value("string"),64 "subject": datasets.Value("string"),65 "message": datasets.Value("string"),66 "lang": datasets.Value("string"),67 "license": datasets.Value("string"),68 "repos": datasets.Value("string"),69# "returncode": datasets.Value("int64"),70# "stderr": datasets.Value("string"),71 }72 ),73 supervised_keys=None,74 citation=_CITATION,75 )76 77 def _split_generators(self, dl_manager):78 79 path_file = dl_manager.download(URL)80 with open(path_file, "r") as f:81 files = json.load(f)[self.config.name]82 83 downloaded_files = dl_manager.download(files)84 return [85 datasets.SplitGenerator(86 name=datasets.Split.TRAIN,87 gen_kwargs={'filepaths': downloaded_files}88 )89 ]90 91 def _generate_examples(self, filepaths):92 """This function returns the examples in the raw (text) form."""93 logger.info("Generating examples from", filepaths)94 95 id_ = 096 for p in filepaths:97 with open(p, "r") as f:98 for row in f:99 data = json.loads(row)100 yield id_, {101 "commit": data["commit"],102 "old_file": data["old_file"],103 "new_file": data["new_file"],104 "old_contents": data["old_contents"],105 "new_contents": data["new_contents"],106 "subject": data["subject"],107 "message": data["message"],108 "lang": data["lang"],109 "license": data["license"],110 "repos": data["repos"], 111 }112 id_ += 1113 