izhx/google-code-jam
Given two codes as the input, the task is to do binary classification (0/1), where 1 stands for semantic equivalence and 0 for others.
036
1from typing import List2import os3import glob4 5import datasets6 7 8_DESCRIPTION = """Given two codes as the input, the task is to do binary classification (0/1), where 1 stands for semantic equivalence and 0 for others."""9 10_CITATION = """@inproceedings{10.1145/3236024.3236068,11author = {Zhao, Gang and Huang, Jeff},12title = {DeepSim: Deep Learning Code Functional Similarity},13year = {2018},14isbn = {9781450355735},15publisher = {Association for Computing Machinery},16address = {New York, NY, USA},17url = {https://doi.org/10.1145/3236024.3236068},18doi = {10.1145/3236024.3236068},19booktitle = {Proceedings of the 2018 26th ACM Joint Meeting on European Software Engineering Conference and Symposium on the Foundations of Software Engineering},20pages = {141–151},21numpages = {11},22keywords = {Classification, Control/Data flow, Code functional similarity, Deep Learning},23location = {Lake Buena Vista, FL, USA},24series = {ESEC/FSE 2018}25}26"""27 28SPLITS = {29 'test': [5, 6, 7, 8, 12], # For test in `Language Models are Universal Embedders` https://arxiv.org/pdf/2310.08232.pdf30 'deepsim': [1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12]31}32_URL = "https://huggingface.co/datasets/izhx/google-code-jam/resolve/main/googlejam4.tar.gz"33 34 35class GoogleCodeJam(datasets.GeneratorBasedBuilder):36 BUILDER_CONFIGS = [37 datasets.BuilderConfig(name='default', version=datasets.Version("1.0.0"), description=_DESCRIPTION)38 ]39 DEFAULT_CONFIG_NAME = "default"40 41 def _info(self):42 return datasets.DatasetInfo(43 description=_DESCRIPTION,44 features=datasets.Features(45 {46 "fn1": datasets.Value("string"),47 "code1": datasets.Value("string"),48 "fn2": datasets.Value("string"),49 "code2": datasets.Value("string"),50 "label": datasets.Value("int32"),51 }52 ),53 homepage="https://github.com/parasol-aser/deepsim",54 citation=_CITATION,55 )56 57 def _split_generators(self, dl_manager: datasets.DownloadManager) -> List[datasets.SplitGenerator]:58 folder = dl_manager.download_and_extract(_URL)59 folder = os.path.join(folder, 'googlejam4_src')60 return [61 datasets.SplitGenerator(name=datasets.Split.TEST, gen_kwargs={"folder": folder, "problems": SPLITS["test"]}),62 datasets.SplitGenerator(name='deepsim', gen_kwargs={"folder": folder, "problems": SPLITS["deepsim"]}),63 ]64 65 def _generate_examples(self, folder, problems: list):66 raw = dict()67 for i in problems:68 group = list()69 for path in sorted(glob.glob(f'{folder}/{i}/*.java')):70 with open(path) as file:71 lines = [l for l in file]72 name = os.path.basename(path)73 group.append((name, ''.join(lines[1:]))) # remove name line74 raw[i] = group75 76 _id = 077 reverse = False78 for i in range(len(problems)):79 vi = raw[problems[i]]80 for n1, (fn1, code1) in enumerate(vi):81 for j in range(i, len(problems)):82 vj = raw[problems[j]]83 match = i == j84 for n2, (fn2, code2) in enumerate(vj):85 if match and n2 <= n1:86 continue87 ins = {'fn1': fn1, 'code1': code1, 'fn2': fn2, 'code2': code2, 'label': int(match)}88 if reverse:89 ins['fn1'], ins['fn2'] = ins['fn2'], ins['fn1']90 ins['code1'], ins['code2'] = ins['code2'], ins['code1']91 yield _id, ins92 _id += 193 reverse = not reverse94 