Team Ai
Datasetpublic

LiZHENGzai/GPRadar-Defect-MultiTask

GPRadar-Defect-MultiTask 数据集 本仓库包含用于微调PaLI-GEMMA多模态模型的地质雷达(GPR)缺陷检测数据集。该数据集专注于地下结构中的空洞和裂缝检测与分析。 数据集结构 数据集组织如下: dataset/ ├── annotations/ - 包含JSON和JSONL格式的标注文件 │ ├── _annotations.train.jsonl - 训练集标注 │ ├── _annotations.valid.jsonl - 验证集标注 │ ├── _annotations.test.jsonl - 测试集标注 │ ├── p-1.v1i.paligemma/ - 主数据集元数据 │ └── p-1.v1i.paligemma-multimodal/ - 多模态数据集元数据 ├── images/ - 包含所有图像文件 特点 包含874张带注释的地质雷达扫描图像 图像预处理为640x640像素大小… See the full description on the dataset page: https://huggingface.co/datasets/LiZHENGzai/GPRadar-Defect-MultiTask.

sourceHugging Facemitupdated 7mo agoView on Hugging Face
0likes8downloads
paligemma_dataset.py308 linesDownload Raw Back to root
1import json2import os3import datasets4 5import jsonlines6 7logger = datasets.logging.get_logger(__name__)8 9_CITATION = """\10@misc{chen2024gpradar,11  title={GPRadar-Defect-MultiTask Dataset},12  author={Chen, Xingqiang},13  year={2024},14  publisher={Hugging Face}15}16"""17 18_DESCRIPTION = """\19GPRadar-Defect-MultiTask Dataset20 21This dataset contains ground penetrating radar (GPR) images and annotations for defect detection and analysis,22designed for training and evaluating multimodal models for GPR defect detection.23The dataset includes both basic defect detection samples and a larger set of24874 annotated images from real-world structural inspections focusing on voids and cracks.25"""26 27_HOMEPAGE = "https://huggingface.co/datasets/xingqiang/GPRadar-Defect-MultiTask"28 29class PaligemmaDataset(datasets.GeneratorBasedBuilder):30    """GPRadar-Defect-MultiTask Dataset for GPR defect detection and analysis."""31 32    VERSION = datasets.Version("1.1.0")33 34    def _info(self):35        return datasets.DatasetInfo(36            description=_DESCRIPTION,37            features=datasets.Features({38                "image": datasets.Image(),39                "boxes": datasets.Sequence(datasets.Sequence(datasets.Value("float32"), length=4)),40                "labels": datasets.Sequence(datasets.ClassLabel(names=["void", "crack"])),41                "caption": datasets.Value("string"),42            }),43            supervised_keys=None,44            homepage=_HOMEPAGE,45            citation=_CITATION,46        )47 48    def _split_generators(self, dl_manager):49        """Returns SplitGenerators."""50        return [51            datasets.SplitGenerator(52                name=datasets.Split.TRAIN,53                gen_kwargs={54                    "split": "train",55                },56            ),57            datasets.SplitGenerator(58                name=datasets.Split.VALIDATION,59                gen_kwargs={60                    "split": "val",61                },62            ),63            datasets.SplitGenerator(64                name=datasets.Split.TEST,65                gen_kwargs={66                    "split": "test",67                },68            ),69        ]70 71    def _generate_examples(self, split):72        """Yields examples."""73        # 统一格式的注释文件74        annotation_file = f"annotations/{split}_unified.json"75        76        if not os.path.exists(annotation_file):77            # 如果统一格式文件不存在,尝试转换78            convert_annotations_to_unified_format()79            80            # 再次检查文件是否已创建81            if not os.path.exists(annotation_file):82                logger.warning(f"找不到统一格式注释文件: {annotation_file},将返回空数据")83                return84 85        # 加载统一格式的注释86        with open(annotation_file, "r", encoding="utf-8") as f:87            annotations = json.load(f)88 89        for idx, ann in enumerate(annotations):90            # 尝试在不同的可能路径中查找图像91            image_found = False92            image_filename = ann["image_filename"]93            94            for image_path in [95                f"images/{split}/{image_filename}",96                f"images/datasets/{image_filename}",97                f"images/{image_filename}",98            ]:99                if os.path.exists(image_path):100                    yield idx, {101                        "image": image_path,102                        "boxes": ann["boxes"],103                        "labels": ann["labels"],104                        "caption": ann["caption"],105                    }106                    image_found = True107                    break108            109            if not image_found:110                logger.warning(f"找不到图像文件: {image_filename},跳过该示例")111 112 113def normalize_image_path(image_path):114    """规范化图像路径,移除多余的前缀"""115    # 处理特殊前缀116    if "p-1.v1i.paligemma-multimodal/dataset/" in image_path:117        return image_path.split("p-1.v1i.paligemma-multimodal/dataset/")[-1]118    return image_path119 120 121def convert_annotations_to_unified_format():122    """将所有注释转换为统一格式"""123    print("开始转换注释为统一格式...")124    125    # 确保annotations目录存在126    os.makedirs("annotations", exist_ok=True)127 128    # 增加对valid分割的处理(有些文件使用valid而不是val)129    for split in ["train", "val", "valid", "test"]:130        print(f"处理 {split} 分割...")131        unified_annotations = []132        133        # 处理 JSON 注释134        json_path = f"annotations/{split}.json"135        print(f"检查 JSON 文件: {json_path}")136        if os.path.exists(json_path):137            print(f"找到 JSON 文件: {json_path}")138            with open(json_path, encoding="utf-8") as f:139                try:140                    annotations = json.load(f)141                    print(f"从 {json_path} 加载了 {len(annotations)} 条注释")142                    for ann in annotations:143                        unified_annotations.append({144                            "image_filename": ann["image_filename"],145                            "boxes": ann["boxes"],146                            "labels": ann["labels"],147                            "caption": ann["caption"],148                            "source": "original"149                        })150                except json.JSONDecodeError:151                    print(f"错误: {json_path} 不是有效的 JSON 文件")152        else:153            print(f"未找到 JSON 文件: {json_path}")154        155        # 查找所有可能的JSONL注释文件156        # 1. 检查根目录157        jsonl_files_to_check = [158            f"_annotations.{split}.jsonl", 159            f"_annotations.{split}1.jsonl"160        ]161        162        # 2. 递归查找子目录中的JSONL文件163        for root, dirs, files in os.walk("annotations"):164            for file in files:165                if file.endswith(f"{split}.jsonl") or file.endswith(f"{split}1.jsonl") or file.endswith(f"{split}2.jsonl"):166                    rel_path = os.path.relpath(os.path.join(root, file), "annotations")167                    if rel_path != file:  # 不是根目录的文件168                        jsonl_files_to_check.append(rel_path)169        170        # 处理所有找到的JSONL文件171        for jsonl_path in jsonl_files_to_check:172            full_path = os.path.join("annotations", jsonl_path)173            print(f"检查 JSONL 文件: {full_path}")174            if os.path.exists(full_path):175                print(f"找到 JSONL 文件: {full_path}")176                annotation_count = 0177                with open(full_path, encoding="utf-8") as f:178                    for line_num, line in enumerate(f, 1):179                        try:180                            line = line.strip()181                            if not line:  # 跳过空行182                                print(f"跳过第 {line_num} 行: 空行")183                                continue184                                185                            ann = json.loads(line)186                            image_filename = ann.get("image", "")187                            188                            if not image_filename:189                                print(f"跳过第 {line_num} 行: 没有图像文件名")190                                continue191                            192                            # 规范化图像路径193                            image_filename = normalize_image_path(image_filename)194                                195                            # 检查图像是否存在196                            image_exists = False197                            possible_image_paths = [198                                f"images/datasets/{image_filename}",199                                f"images/train/{image_filename}",200                                f"images/val/{image_filename}",201                                f"images/test/{image_filename}",202                                f"images/{image_filename}"  # 直接在images目录下203                            ]204                            205                            for img_path in possible_image_paths:206                                if os.path.exists(img_path):207                                    image_exists = True208                                    break209                            210                            if not image_exists:211                                print(f"警告: 图像文件不存在: {image_filename}")212                                continue213                            214                            # 转换为统一格式215                            if "annotations" in ann:216                                # 处理新格式217                                boxes = [[b["x"], b["y"], b["width"], b["height"]] for b in ann["annotations"]]218                                labels = [0 if b["class"] == "void" else 1 for b in ann["annotations"]]219                                caption = f"Image contains {len(boxes)} defects: " + \220                                        ", ".join([b["class"] for b in ann["annotations"]])221                            else:222                                # 处理旧格式 (prefix/suffix)223                                boxes = []224                                labels = []225                                caption = ann.get("prefix", "")226                                227                                if "suffix" in ann:228                                    parts = ann["suffix"].split()229                                    for i, part in enumerate(parts):230                                        if "<loc" in part:231                                            # 解析位置232                                            coords = []233                                            loc_str = part234                                            while loc_str.startswith("<loc") and len(coords) < 4:235                                                try:236                                                    # 提取坐标237                                                    coord_value = int(loc_str[4:loc_str.find(">")])238                                                    coords.append(coord_value / 1024)  # 归一化坐标239                                                    # 移除已处理的部分240                                                    loc_str = loc_str[loc_str.find(">")+1:]241                                                except (ValueError, IndexError):242                                                    break243                                            244                                            if len(coords) == 4:245                                                boxes.append(coords)246                                                # 查找标签(通常在下一个部分)247                                                label_idx = 1248                                                while i + label_idx < len(parts) and not parts[i + label_idx].startswith("<loc"):249                                                    label_text = parts[i + label_idx]250                                                    if "void" in label_text:251                                                        labels.append(0)252                                                        break253                                                    elif "crack" in label_text:254                                                        labels.append(1)255                                                        break256                                                    label_idx += 1257                                                258                                                # 如果未找到特定标签,默认为void259                                                if len(labels) < len(boxes):260                                                    labels.append(0)261                            262                            unified_annotations.append({263                                "image_filename": image_filename,264                                "boxes": boxes,265                                "labels": labels,266                                "caption": caption,267                                "source": "p1v1"268                            })269                            annotation_count += 1270                        except json.JSONDecodeError as e:271                            print(f"警告: {full_path} 第 {line_num} 行不是有效的 JSON: {e}")272                            continue273                print(f"从 {full_path} 加载了 {annotation_count} 条注释")274            else:275                print(f"未找到 JSONL 文件: {full_path}")276        277        # 如果是valid分割,与val合并278        if split == "valid":279            val_annotations = []280            if os.path.exists(f"annotations/val_unified.json"):281                try:282                    with open(f"annotations/val_unified.json", "r", encoding="utf-8") as f:283                        val_annotations = json.load(f)284                    print(f"加载现有val分割注释,共 {len(val_annotations)} 条记录")285                    286                    # 合并注释,避免重复287                    existing_filenames = {ann["image_filename"] for ann in val_annotations}288                    for ann in unified_annotations:289                        if ann["image_filename"] not in existing_filenames:290                            val_annotations.append(ann)291                            existing_filenames.add(ann["image_filename"])292                    293                    print(f"将valid分割与val分割合并,共 {len(val_annotations)} 条记录")294                    unified_annotations = val_annotations295                except Exception as e:296                    print(f"合并valid和val分割时出错: {e}")297                    298        # 保存统一格式的注释299        if unified_annotations:300            # 对于valid分割,保存为val_unified.json301            save_split = "val" if split == "valid" else split302            print(f"为 {save_split} 创建统一格式注释,共 {len(unified_annotations)} 条记录")303            unified_path = f"annotations/{save_split}_unified.json"304            with open(unified_path, "w", encoding="utf-8") as f:305                json.dump(unified_annotations, f, ensure_ascii=False, indent=2)306            print(f"已保存统一格式注释到: {unified_path}")307        else:308            print(f"警告: {split} 没有有效的注释,跳过创建统一格式文件")