Team Ai
Datasetpublic

bigcode/commitpack

CommitPack is is a 4TB dataset of commits scraped from GitHub repositories that are permissively licensed.

sourceHugging Facemitupdated 2y agoView on Hugging Face
79likes15kdownloads
commitpack.py113 linesDownload Raw Back to root
1"""CommitPack"""2 3import json4import datasets5 6 7logger = datasets.logging.get_logger(__name__)8 9### To create paths ###10def get_paths():11    import json, glob, os12    files = {}13    for lang_dir in os.listdir("./data"):14        print("Processing", lang_dir)15        if not os.path.isdir("data/" + lang_dir):16            print(f"Skipping {lang_dir} as it is not a directory")17            continue18        for file in glob.glob(f"data/{lang_dir}/*.jsonl"):19            files[lang_dir] = files.get(lang_dir, []) + [file]20    with open(f"paths.json", "w") as f:21        json.dump(files, f)22    return files23 24_CITATION = """\25@article{muennighoff2023octopack,26      title={OctoPack: Instruction Tuning Code Large Language Models}, 27      author={Niklas Muennighoff and Qian Liu and Armel Zebaze and Qinkai Zheng and Binyuan Hui and Terry Yue Zhuo and Swayam Singh and Xiangru Tang and Leandro von Werra and Shayne Longpre},28      journal={arXiv preprint arXiv:2308.07124},29      year={2023}30}31"""32 33_DESCRIPTION = """\34CommitPack is is a 4TB dataset of commits scraped from GitHub repositories that are permissively licensed.35"""36 37URL = "https://huggingface.co/datasets/bigcode/commitpack/resolve/main/paths.json"38 39_LANG = ["json", "xml", "text", "javascript", "objective-c++", "python", "c", "c++", "markdown", "java", "html", "yaml", "go", "csv", "php", "jupyter-notebook", "gettext-catalog", "sql", "unity3d-asset", "typescript", "web-ontology-language", "ruby", "c#", "nix", "shell", "perl", "tex", "css", "restructuredtext", "rust", "groff", "ini", "scala", "coffeescript", "haskell", "swift", "lua", "svg", "gas", "ocaml", "erlang", "makefile", "asciidoc", "emacs-lisp", "scss", "clojure", "org", "common-lisp", "diff", "groovy", "html+erb", "nesc", "dart", "powershell", "f#", "dm", "kotlin", "pascal", "jsx", "viml", "actionscript", "cython", "turtle", "less", "mathematica", "xslt", "scheme", "perl6", "edn", "fortran", "java-server-pages", "standard-ml", "cmake", "json5", "vala", "vue", "freemarker", "graphql", "twig", "tcl", "pod", "dockerfile", "yacc", "postscript", "racket", "eagle", "haxe", "julia", "handlebars", "smarty", "visual-basic", "literate-haskell", "smalltalk", "isabelle", "nimrod", "zig", "m4", "max", "elixir", "mako", "arduino", "jade", "haml", "elm", "purebasic", "coldfusion", "lean", "r", "cuda", "textile", "robotframework", "abap", "rdoc", "llvm", "ada", "batchfile", "qml", "jasmin", "assembly", "g-code", "cucumber", "html+php", "kicad", "api-blueprint", "eiffel", "toml", "modelica", "bitbake", "lex", "stylus", "protocol-buffer", "unknown", "nit", "factor", "xs", "sass", "parrot-internal-representation", "html+django", "mediawiki", "logos", "genshi", "coldfusion-cfc", "xtend", "sqf", "vhdl", "antlr", "systemverilog", "hcl", "asp", "nsis", "inform-7", "slim", "groovy-server-pages", "ceylon", "fish", "processing", "component-pascal", "lasso", "glsl", "saltstack", "xbase", "autohotkey", "liquid", "purescript", "agda", "inno-setup", "oz", "chapel", "arc", "opencl", "graphviz-dot", "pawn", "jsoniq", "bluespec", "smali", "krl", "maple", "unrealscript", "ooc", "pure-data", "xquery", "digital-command-language", "moonscript", "awk", "pike", "livescript", "solidity", "monkey", "jsonld", "zephir", "crystal", "rhtml", "stata", "idris", "raml", "openscad", "red", "c2hs-haskell", "cycript", "applescript", "mupad", "literate-agda", "boo", "sourcepawn", "qmake", "ragel-in-ruby-host", "io", "desktop", "propeller-spin", "thrift", "volt", "xproc", "igor-pro", "lolcode", "html+eex", "logtalk", "mirah", "gnuplot", "literate-coffeescript", "jflex", "emberscript", "cobol", "yang", "rebol", "linker-script", "cartocss", "urweb", "rmarkdown", "darcs-patch", "csound", "squirrel", "apl", "hlsl", "latte", "pony", "ioke", "hy", "uno", "pan", "xojo", "papyrus", "stan", "slash", "supercollider", "vcl", "smt", "glyph", "wisp", "renpy", "clips", "dns-zone", "sas", "rouge", "ec", "dylan", "tcsh", "aspectj", "netlogo", "gap", "fancy", "coq", "click", "capn-proto", "flux", "forth", "ats", "netlinx", "clean", "parrot-assembly", "alloy", "lfe", "gdscript", "augeas", "sparql", "lilypond", "scilab", "autoit", "myghty", "blitzmax", "creole", "harbour", "piglatin", "opa", "sage", "ston", "maxscript", "lsl", "gentoo-ebuild", "nu", "bro", "xc", "j", "metal", "module-management-system", "webidl", "tea", "redcode", "shen", "pov-ray-sdl", "x10", "brainfuck", "ninja", "golo", "webassembly", "self", "labview", "octave", "pogoscript", "d", "http", "ecl", "chuck", "gosu", "parrot", "opal", "objective-j", "kit", "gams", "prolog", "clarion", "mask", "brightscript", "scaml", "matlab", "idl", "ags-script", "lookml", "apacheconf", "oxygene", "txl", "grammatical-framework", "renderscript", "mtml", "unified-parallel-c", "dogescript", "gentoo-eclass", "zimpl", "irc-log", "fantom", "numpy", "cirru", "xpages", "nginx", "objdump", "python-traceback", "realbasic", "befunge", "bison", "m", "omgrofl"]40_LICENSE = "Apache License 2.0"41_VERSION = datasets.Version("1.0.0", "")42 43 44class CommitPack(datasets.GeneratorBasedBuilder):45    BUILDER_CONFIGS = [46        datasets.BuilderConfig(47            name=lang,48            description=f"CommitPack {lang}",49            version=_VERSION,50        )51        for lang in _LANG52    ]53 54    def _info(self):55        return datasets.DatasetInfo(56            description=_DESCRIPTION,57            features=datasets.Features(58                {59                    "commit": datasets.Value("string"),60                    "old_file": datasets.Value("string"),61                    "new_file": datasets.Value("string"),62                    "old_contents": datasets.Value("string"),63                    "new_contents": datasets.Value("string"),64                    "subject": datasets.Value("string"),65                    "message": datasets.Value("string"),66                    "lang": datasets.Value("string"),67                    "license": datasets.Value("string"),68                    "repos": datasets.Value("string"),69#                    "returncode": datasets.Value("int64"),70#                    "stderr": datasets.Value("string"),71                }72            ),73            supervised_keys=None,74            citation=_CITATION,75        )76    77    def _split_generators(self, dl_manager):78 79        path_file = dl_manager.download(URL)80        with open(path_file, "r") as f:81            files = json.load(f)[self.config.name]82 83        downloaded_files = dl_manager.download(files)84        return [85            datasets.SplitGenerator(86                name=datasets.Split.TRAIN,87                gen_kwargs={'filepaths': downloaded_files}88            )89        ]90 91    def _generate_examples(self, filepaths):92        """This function returns the examples in the raw (text) form."""93        logger.info("Generating examples from", filepaths)94 95        id_ = 096        for p in filepaths:97            with open(p, "r") as f:98                for row in f:99                    data = json.loads(row)100                    yield id_, {101                        "commit": data["commit"],102                        "old_file": data["old_file"],103                        "new_file": data["new_file"],104                        "old_contents": data["old_contents"],105                        "new_contents": data["new_contents"],106                        "subject": data["subject"],107                        "message": data["message"],108                        "lang": data["lang"],109                        "license": data["license"],110                        "repos": data["repos"],                    111                    }112                    id_ += 1113