GEM/DatasetCardForm
9
1import argparse2import json3import pathlib4import os5 6parser = argparse.ArgumentParser(7 description="Format the output of the data card tool as .md for the hub."8)9parser.add_argument("--input_path", "-i", type=pathlib.Path, required=False)10parser.add_argument("--output_path", "-o", type=pathlib.Path, required=False)11args = parser.parse_args()12 13 14def read_json_file(json_path: pathlib.Path):15 """Load a json file and return it as object."""16 with open(json_path, "r") as f:17 data = json.load(f)18 return data19 20 21def save_file(json_path: pathlib.Path, json_obj: str):22 """Takes a string and saves it as .md file."""23 with open(json_path, "w") as f:24 f.write(json.dumps(json_obj, indent=2))25 26 27def construct_json(dataset_name: str, data_card_data: dict, text_by_key: dict):28 """Constructs the json file29 30 This function iterates through text_by_key and extracts all answers from31 the data_card_data object. It uses the levels of hierarchy as indicator for32 the heading indentation and does not change the order in which anything33 appears.34 35 Args:36 data_card_data: Output from the data card tool37 text_by_key: configuration defined in key_to_question.json38 39 Returns:40 data_card_md_string: json content41 """42 43 try:44 website_link = data_card_data["overview"]["where"]["website"]45 except KeyError:46 website_link = ""47 try:48 paper_link = data_card_data["overview"]["where"]["paper-url"]49 except KeyError:50 paper_link = ""51 try:52 authors = data_card_data["overview"]["credit"]["creators"]53 except KeyError:54 authors = ""55 try:56 summary = data_card_data["overview"]["what"]["dataset"]57 except KeyError:58 summary = "Placeholder"59 60 61 # Add summary blurb with loading script and link to GEM loader62 summary +=f"\n\nYou can load the dataset via:\n```\nimport datasets\ndata = datasets.load_dataset('GEM/{dataset_name}')\n```\nThe data loader can be found [here](https://huggingface.co/datasets/GEM/{dataset_name})."63 64 new_json = {65 "name": dataset_name,66 "summary": summary,67 "sections": [68 ],69 }70 71 if website_link:72 new_json["website"] = website_link73 if paper_link:74 new_json["paper"] = paper_link75 if authors:76 new_json["authors"] = authors77 78 79 total_questions = 080 total_words = 081 82 for main_key, main_content in text_by_key.items():83 l2_data = {84 "title": main_content["section-title"],85 "level": 2,86 "subsections": []87 }88 if main_key not in data_card_data:89 continue90 for second_key, second_content in main_content.items():91 if second_key == "section-title":92 continue93 # Skip summary data since it is already in the header.94 if main_key == "overview" and second_key == "what":95 continue96 l3_data = {97 "title": second_content["section-title"],98 "level": 3,99 "fields": []100 }101 for final_key, final_content in second_content.items():102 if final_key == "section-title":103 continue104 try:105 total_questions += 1106 answer = data_card_data[main_key][second_key].get(final_key, "N/A")107 except:108 # print(main_key, second_key, final_key)109 # print("==="*50)110 # print(data_card_data)111 continue112 # Skip empty answers.113 if isinstance(answer, str):114 if answer.lower() == "n/a":115 continue116 if not answer:117 continue118 119 if isinstance(answer, list):120 answer = ", ".join([f"`{a}`" for a in answer])121 122 json_answer = {123 "title": final_content["title"],124 "level": 4,125 "content": answer,126 "flags": final_content["flags"],127 "info": final_content["info"],128 "scope": final_content["scope"],129 }130 total_words += len(answer.split())131 l3_data["fields"].append(json_answer)132 l2_data["subsections"].append(l3_data)133 new_json["sections"].append(l2_data)134 print(f"Total questions {total_questions}")135 print(f"total words: {total_words}")136 return new_json, total_words137 138 139 140 141if __name__ == "__main__":142 143 text_by_key = read_json_file(144 os.path.join(os.path.dirname(__file__), "key_to_question.json")145 )146 total_words_across_everything = 0147 for dataset in os.listdir("../../../GEMv2"):148 data_card_path = f"../../../GEMv2/{dataset}/{dataset}.json"149 if os.path.exists(data_card_path):150 print(f"Now processing {dataset}.")151 new_path = f"datacards/{dataset}.json"152 data_card_data = read_json_file(data_card_path)153 data_card_json, total_cur_words = construct_json(dataset, data_card_data, text_by_key)154 total_words_across_everything += total_cur_words155 156 save_file(new_path, data_card_json)157 else:158 print(f"{dataset} has no data card!")159 print(total_words_across_everything)160 # data_card_json = construct_json(data_card_data, text_by_key)161 # save_file(args.output_path, data_card_json)162 