Team Ai
Datasetpublic

intelli-zen/cppe-5

CPPE - 5 (Medical Personal Protective Equipment) is a new challenging dataset with the goal to allow the study of subordinate categorization of medical personal protective equipments, which is not possible with other popular data sets that focus on broad level categories.

sourceHugging Faceapache-2.0updated 3y agoView on Hugging Face
0likes712downloads
make_jsonl.py65 linesDownload Raw Back to examples
1#!/usr/bin/python32# -*- coding: utf-8 -*-3from collections import defaultdict4import json5import os6from pathlib import Path7 8from project_settings import project_path9 10 11def main():12 13    for subset in ["train", "test"]:14        filename = project_path / "data/annotations/{}.json".format(subset)15        with open(filename.as_posix(), "r", encoding="utf-8") as f:16            js = json.load(f)17 18        images = js["images"]19        type_ = js["type"]20        annotations = js["annotations"]21        categories = js["categories"]22 23        index_to_label = dict()24        for category in categories:25            index = category["id"]26            name = category["name"]27            index_to_label[index] = name28 29        # print(images)30        image_id_to_annotations = defaultdict(list)31        for annotation in annotations:32            image_id = annotation["image_id"]33            image_id_to_annotations[image_id].append(annotation)34 35        to_filename = project_path / "data/annotations/{}.jsonl".format(subset)36        with open(to_filename.as_posix(), "w", encoding="utf-8") as f:37            for image in images:38                image_id = image["id"]39                annotations = image_id_to_annotations[image_id]40 41                image_path = Path("data/images") / image["file_name"]42 43                row = {44                    "image_id": image["id"],45                    "image": image_path.as_posix(),46                    "width": image["width"],47                    "height": image["height"],48                    "objects": [49                        {50                            "id": annotation["id"],51                            "area": annotation["area"],52                            "bbox": annotation["bbox"],53                            "category": index_to_label[annotation["category_id"]],54                        } for annotation in annotations55                    ]56                }57                row = json.dumps(row, ensure_ascii=False)58                f.write("{}\n".format(row))59 60    return61 62 63if __name__ == '__main__':64    main()65