Team Ai
Datasetpublic

bleugreen/typescript-chunks

typescript-chunks A dataset of TypeScript snippets, processed from the typescript subset of the-stack-smol. Processing Each source file is parsed with the TypeScript AST and queried for 'semantic chunks' of the following types. FunctionDeclaration ---- 8205 ArrowFunction --------- 33890 ClassDeclaration ------- 5325 InterfaceDeclaration -- 12884 EnumDeclaration --------- 518 TypeAliasDeclaration --- 3580 MethodDeclaration ----- 24713 Leading comments are added… See the full description on the dataset page: https://huggingface.co/datasets/bleugreen/typescript-chunks.

sourceHugging Faceupdated 3y agoView on Hugging Face
3likes34downloads
parse.py46 linesDownload Raw Back to root
1import subprocess2from datasets import load_dataset, Dataset3import json 4from tqdm import tqdm5 6ds = load_dataset("bigcode/the-stack-smol", data_dir='data/typescript')7 8def split_ts_into_chunks(ts_code):9    result = subprocess.run(10        ['node', 'parse_ts.js'],11        input=ts_code,12        text=True,13    )14 15    if result.returncode != 0:16        raise Exception('Error in TypeScript parsing')17 18    with open('semantic_chunks.jsonl', 'r') as file:19        lines = file.read().splitlines()20 21    chunks = [json.loads(line) for line in lines]22    with open('semantic_chunks.jsonl', 'w'):23        pass24 25    return chunks26 27 28def chunk_ts_file(data):29    funcs = split_ts_into_chunks(data['content'])30    for i in range(len(funcs)):31        funcs[i]['repo'] = data['repository_name']32        funcs[i]['path'] = data['path']33        funcs[i]['language'] = data['lang']34    return funcs35 36chunks = []37for i in tqdm(range(len(ds['train']))):38    chunk = chunk_ts_file(ds['train'][i])39    chunks +=(chunk)40    if i%100 == 0:41        print(len(chunks))42 43dataset = Dataset.from_list(chunks)44print(dataset)45dataset.to_json('ts-chunks.json')46