Team Ai
Apppublic

GEM/DatasetCardForm

sourceHugging Faceupdated 4y agoView on Hugging Face
9likes
reformat_json.py162 linesDownload Raw Back to formatting
1import argparse2import json3import pathlib4import os5 6parser = argparse.ArgumentParser(7    description="Format the output of the data card tool as .md for the hub."8)9parser.add_argument("--input_path", "-i", type=pathlib.Path, required=False)10parser.add_argument("--output_path", "-o", type=pathlib.Path, required=False)11args = parser.parse_args()12 13 14def read_json_file(json_path: pathlib.Path):15    """Load a json file and return it as object."""16    with open(json_path, "r") as f:17        data = json.load(f)18    return data19 20 21def save_file(json_path: pathlib.Path, json_obj: str):22    """Takes a string and saves it as .md file."""23    with open(json_path, "w") as f:24        f.write(json.dumps(json_obj, indent=2))25 26 27def construct_json(dataset_name: str, data_card_data: dict, text_by_key: dict):28  """Constructs the json file29 30  This function iterates through text_by_key and extracts all answers from31  the data_card_data object. It uses the levels of hierarchy as indicator for32  the heading indentation and does not change the order in which anything33  appears.34 35  Args:36      data_card_data: Output from the data card tool37      text_by_key: configuration defined in key_to_question.json38 39  Returns:40      data_card_md_string: json content41  """42 43  try:44    website_link = data_card_data["overview"]["where"]["website"]45  except KeyError:46    website_link = ""47  try:48    paper_link = data_card_data["overview"]["where"]["paper-url"]49  except KeyError:50    paper_link = ""51  try:52    authors = data_card_data["overview"]["credit"]["creators"]53  except KeyError:54    authors = ""55  try:56    summary = data_card_data["overview"]["what"]["dataset"]57  except KeyError:58    summary = "Placeholder"59 60 61  # Add summary blurb with loading script and link to GEM loader62  summary +=f"\n\nYou can load the dataset via:\n```\nimport datasets\ndata = datasets.load_dataset('GEM/{dataset_name}')\n```\nThe data loader can be found [here](https://huggingface.co/datasets/GEM/{dataset_name})."63 64  new_json = {65      "name": dataset_name,66      "summary": summary,67      "sections": [68      ],69  }70 71  if website_link:72    new_json["website"] = website_link73  if paper_link:74    new_json["paper"] = paper_link75  if authors:76    new_json["authors"] = authors77 78 79  total_questions = 080  total_words = 081 82  for main_key, main_content in text_by_key.items():83    l2_data = {84              "title": main_content["section-title"],85              "level": 2,86              "subsections": []87    }88    if main_key not in data_card_data:89      continue90    for second_key, second_content in main_content.items():91      if second_key == "section-title":92        continue93      # Skip summary data since it is already in the header.94      if main_key == "overview" and second_key == "what":95        continue96      l3_data = {97                      "title": second_content["section-title"],98                      "level": 3,99                      "fields": []100      }101      for final_key, final_content in second_content.items():102        if final_key == "section-title":103          continue104        try:105          total_questions += 1106          answer = data_card_data[main_key][second_key].get(final_key, "N/A")107        except:108          # print(main_key, second_key, final_key)109          # print("==="*50)110          # print(data_card_data)111          continue112        # Skip empty answers.113        if isinstance(answer, str):114          if answer.lower() == "n/a":115            continue116        if not answer:117          continue118 119        if isinstance(answer, list):120          answer = ", ".join([f"`{a}`" for a in answer])121 122        json_answer = {123          "title": final_content["title"],124          "level": 4,125          "content": answer,126          "flags": final_content["flags"],127          "info": final_content["info"],128          "scope": final_content["scope"],129        }130        total_words += len(answer.split())131        l3_data["fields"].append(json_answer)132      l2_data["subsections"].append(l3_data)133    new_json["sections"].append(l2_data)134  print(f"Total questions {total_questions}")135  print(f"total words: {total_words}")136  return new_json, total_words137 138 139 140 141if __name__ == "__main__":142 143  text_by_key = read_json_file(144      os.path.join(os.path.dirname(__file__), "key_to_question.json")145  )146  total_words_across_everything = 0147  for dataset in os.listdir("../../../GEMv2"):148    data_card_path = f"../../../GEMv2/{dataset}/{dataset}.json"149    if os.path.exists(data_card_path):150      print(f"Now processing {dataset}.")151      new_path = f"datacards/{dataset}.json"152      data_card_data = read_json_file(data_card_path)153      data_card_json, total_cur_words = construct_json(dataset, data_card_data, text_by_key)154      total_words_across_everything += total_cur_words155 156      save_file(new_path, data_card_json)157    else:158      print(f"{dataset} has no data card!")159  print(total_words_across_everything)160  # data_card_json = construct_json(data_card_data, text_by_key)161  # save_file(args.output_path, data_card_json)162