bleugreen/typescript-chunks
typescript-chunks A dataset of TypeScript snippets, processed from the typescript subset of the-stack-smol. Processing Each source file is parsed with the TypeScript AST and queried for 'semantic chunks' of the following types. FunctionDeclaration ---- 8205 ArrowFunction --------- 33890 ClassDeclaration ------- 5325 InterfaceDeclaration -- 12884 EnumDeclaration --------- 518 TypeAliasDeclaration --- 3580 MethodDeclaration ----- 24713 Leading comments are added… See the full description on the dataset page: https://huggingface.co/datasets/bleugreen/typescript-chunks.
334
1import subprocess2from datasets import load_dataset, Dataset3import json 4from tqdm import tqdm5 6ds = load_dataset("bigcode/the-stack-smol", data_dir='data/typescript')7 8def split_ts_into_chunks(ts_code):9 result = subprocess.run(10 ['node', 'parse_ts.js'],11 input=ts_code,12 text=True,13 )14 15 if result.returncode != 0:16 raise Exception('Error in TypeScript parsing')17 18 with open('semantic_chunks.jsonl', 'r') as file:19 lines = file.read().splitlines()20 21 chunks = [json.loads(line) for line in lines]22 with open('semantic_chunks.jsonl', 'w'):23 pass24 25 return chunks26 27 28def chunk_ts_file(data):29 funcs = split_ts_into_chunks(data['content'])30 for i in range(len(funcs)):31 funcs[i]['repo'] = data['repository_name']32 funcs[i]['path'] = data['path']33 funcs[i]['language'] = data['lang']34 return funcs35 36chunks = []37for i in tqdm(range(len(ds['train']))):38 chunk = chunk_ts_file(ds['train'][i])39 chunks +=(chunk)40 if i%100 == 0:41 print(len(chunks))42 43dataset = Dataset.from_list(chunks)44print(dataset)45dataset.to_json('ts-chunks.json')46 