LiZHENGzai/GPRadar-Defect-MultiTask
GPRadar-Defect-MultiTask 数据集 本仓库包含用于微调PaLI-GEMMA多模态模型的地质雷达(GPR)缺陷检测数据集。该数据集专注于地下结构中的空洞和裂缝检测与分析。 数据集结构 数据集组织如下: dataset/ ├── annotations/ - 包含JSON和JSONL格式的标注文件 │ ├── _annotations.train.jsonl - 训练集标注 │ ├── _annotations.valid.jsonl - 验证集标注 │ ├── _annotations.test.jsonl - 测试集标注 │ ├── p-1.v1i.paligemma/ - 主数据集元数据 │ └── p-1.v1i.paligemma-multimodal/ - 多模态数据集元数据 ├── images/ - 包含所有图像文件 特点 包含874张带注释的地质雷达扫描图像 图像预处理为640x640像素大小… See the full description on the dataset page: https://huggingface.co/datasets/LiZHENGzai/GPRadar-Defect-MultiTask.
08
1import json2import os3import datasets4 5import jsonlines6 7logger = datasets.logging.get_logger(__name__)8 9_CITATION = """\10@misc{chen2024gpradar,11 title={GPRadar-Defect-MultiTask Dataset},12 author={Chen, Xingqiang},13 year={2024},14 publisher={Hugging Face}15}16"""17 18_DESCRIPTION = """\19GPRadar-Defect-MultiTask Dataset20 21This dataset contains ground penetrating radar (GPR) images and annotations for defect detection and analysis,22designed for training and evaluating multimodal models for GPR defect detection.23The dataset includes both basic defect detection samples and a larger set of24874 annotated images from real-world structural inspections focusing on voids and cracks.25"""26 27_HOMEPAGE = "https://huggingface.co/datasets/xingqiang/GPRadar-Defect-MultiTask"28 29class PaligemmaDataset(datasets.GeneratorBasedBuilder):30 """GPRadar-Defect-MultiTask Dataset for GPR defect detection and analysis."""31 32 VERSION = datasets.Version("1.1.0")33 34 def _info(self):35 return datasets.DatasetInfo(36 description=_DESCRIPTION,37 features=datasets.Features({38 "image": datasets.Image(),39 "boxes": datasets.Sequence(datasets.Sequence(datasets.Value("float32"), length=4)),40 "labels": datasets.Sequence(datasets.ClassLabel(names=["void", "crack"])),41 "caption": datasets.Value("string"),42 }),43 supervised_keys=None,44 homepage=_HOMEPAGE,45 citation=_CITATION,46 )47 48 def _split_generators(self, dl_manager):49 """Returns SplitGenerators."""50 return [51 datasets.SplitGenerator(52 name=datasets.Split.TRAIN,53 gen_kwargs={54 "split": "train",55 },56 ),57 datasets.SplitGenerator(58 name=datasets.Split.VALIDATION,59 gen_kwargs={60 "split": "val",61 },62 ),63 datasets.SplitGenerator(64 name=datasets.Split.TEST,65 gen_kwargs={66 "split": "test",67 },68 ),69 ]70 71 def _generate_examples(self, split):72 """Yields examples."""73 # 统一格式的注释文件74 annotation_file = f"annotations/{split}_unified.json"75 76 if not os.path.exists(annotation_file):77 # 如果统一格式文件不存在,尝试转换78 convert_annotations_to_unified_format()79 80 # 再次检查文件是否已创建81 if not os.path.exists(annotation_file):82 logger.warning(f"找不到统一格式注释文件: {annotation_file},将返回空数据")83 return84 85 # 加载统一格式的注释86 with open(annotation_file, "r", encoding="utf-8") as f:87 annotations = json.load(f)88 89 for idx, ann in enumerate(annotations):90 # 尝试在不同的可能路径中查找图像91 image_found = False92 image_filename = ann["image_filename"]93 94 for image_path in [95 f"images/{split}/{image_filename}",96 f"images/datasets/{image_filename}",97 f"images/{image_filename}",98 ]:99 if os.path.exists(image_path):100 yield idx, {101 "image": image_path,102 "boxes": ann["boxes"],103 "labels": ann["labels"],104 "caption": ann["caption"],105 }106 image_found = True107 break108 109 if not image_found:110 logger.warning(f"找不到图像文件: {image_filename},跳过该示例")111 112 113def normalize_image_path(image_path):114 """规范化图像路径,移除多余的前缀"""115 # 处理特殊前缀116 if "p-1.v1i.paligemma-multimodal/dataset/" in image_path:117 return image_path.split("p-1.v1i.paligemma-multimodal/dataset/")[-1]118 return image_path119 120 121def convert_annotations_to_unified_format():122 """将所有注释转换为统一格式"""123 print("开始转换注释为统一格式...")124 125 # 确保annotations目录存在126 os.makedirs("annotations", exist_ok=True)127 128 # 增加对valid分割的处理(有些文件使用valid而不是val)129 for split in ["train", "val", "valid", "test"]:130 print(f"处理 {split} 分割...")131 unified_annotations = []132 133 # 处理 JSON 注释134 json_path = f"annotations/{split}.json"135 print(f"检查 JSON 文件: {json_path}")136 if os.path.exists(json_path):137 print(f"找到 JSON 文件: {json_path}")138 with open(json_path, encoding="utf-8") as f:139 try:140 annotations = json.load(f)141 print(f"从 {json_path} 加载了 {len(annotations)} 条注释")142 for ann in annotations:143 unified_annotations.append({144 "image_filename": ann["image_filename"],145 "boxes": ann["boxes"],146 "labels": ann["labels"],147 "caption": ann["caption"],148 "source": "original"149 })150 except json.JSONDecodeError:151 print(f"错误: {json_path} 不是有效的 JSON 文件")152 else:153 print(f"未找到 JSON 文件: {json_path}")154 155 # 查找所有可能的JSONL注释文件156 # 1. 检查根目录157 jsonl_files_to_check = [158 f"_annotations.{split}.jsonl", 159 f"_annotations.{split}1.jsonl"160 ]161 162 # 2. 递归查找子目录中的JSONL文件163 for root, dirs, files in os.walk("annotations"):164 for file in files:165 if file.endswith(f"{split}.jsonl") or file.endswith(f"{split}1.jsonl") or file.endswith(f"{split}2.jsonl"):166 rel_path = os.path.relpath(os.path.join(root, file), "annotations")167 if rel_path != file: # 不是根目录的文件168 jsonl_files_to_check.append(rel_path)169 170 # 处理所有找到的JSONL文件171 for jsonl_path in jsonl_files_to_check:172 full_path = os.path.join("annotations", jsonl_path)173 print(f"检查 JSONL 文件: {full_path}")174 if os.path.exists(full_path):175 print(f"找到 JSONL 文件: {full_path}")176 annotation_count = 0177 with open(full_path, encoding="utf-8") as f:178 for line_num, line in enumerate(f, 1):179 try:180 line = line.strip()181 if not line: # 跳过空行182 print(f"跳过第 {line_num} 行: 空行")183 continue184 185 ann = json.loads(line)186 image_filename = ann.get("image", "")187 188 if not image_filename:189 print(f"跳过第 {line_num} 行: 没有图像文件名")190 continue191 192 # 规范化图像路径193 image_filename = normalize_image_path(image_filename)194 195 # 检查图像是否存在196 image_exists = False197 possible_image_paths = [198 f"images/datasets/{image_filename}",199 f"images/train/{image_filename}",200 f"images/val/{image_filename}",201 f"images/test/{image_filename}",202 f"images/{image_filename}" # 直接在images目录下203 ]204 205 for img_path in possible_image_paths:206 if os.path.exists(img_path):207 image_exists = True208 break209 210 if not image_exists:211 print(f"警告: 图像文件不存在: {image_filename}")212 continue213 214 # 转换为统一格式215 if "annotations" in ann:216 # 处理新格式217 boxes = [[b["x"], b["y"], b["width"], b["height"]] for b in ann["annotations"]]218 labels = [0 if b["class"] == "void" else 1 for b in ann["annotations"]]219 caption = f"Image contains {len(boxes)} defects: " + \220 ", ".join([b["class"] for b in ann["annotations"]])221 else:222 # 处理旧格式 (prefix/suffix)223 boxes = []224 labels = []225 caption = ann.get("prefix", "")226 227 if "suffix" in ann:228 parts = ann["suffix"].split()229 for i, part in enumerate(parts):230 if "<loc" in part:231 # 解析位置232 coords = []233 loc_str = part234 while loc_str.startswith("<loc") and len(coords) < 4:235 try:236 # 提取坐标237 coord_value = int(loc_str[4:loc_str.find(">")])238 coords.append(coord_value / 1024) # 归一化坐标239 # 移除已处理的部分240 loc_str = loc_str[loc_str.find(">")+1:]241 except (ValueError, IndexError):242 break243 244 if len(coords) == 4:245 boxes.append(coords)246 # 查找标签(通常在下一个部分)247 label_idx = 1248 while i + label_idx < len(parts) and not parts[i + label_idx].startswith("<loc"):249 label_text = parts[i + label_idx]250 if "void" in label_text:251 labels.append(0)252 break253 elif "crack" in label_text:254 labels.append(1)255 break256 label_idx += 1257 258 # 如果未找到特定标签,默认为void259 if len(labels) < len(boxes):260 labels.append(0)261 262 unified_annotations.append({263 "image_filename": image_filename,264 "boxes": boxes,265 "labels": labels,266 "caption": caption,267 "source": "p1v1"268 })269 annotation_count += 1270 except json.JSONDecodeError as e:271 print(f"警告: {full_path} 第 {line_num} 行不是有效的 JSON: {e}")272 continue273 print(f"从 {full_path} 加载了 {annotation_count} 条注释")274 else:275 print(f"未找到 JSONL 文件: {full_path}")276 277 # 如果是valid分割,与val合并278 if split == "valid":279 val_annotations = []280 if os.path.exists(f"annotations/val_unified.json"):281 try:282 with open(f"annotations/val_unified.json", "r", encoding="utf-8") as f:283 val_annotations = json.load(f)284 print(f"加载现有val分割注释,共 {len(val_annotations)} 条记录")285 286 # 合并注释,避免重复287 existing_filenames = {ann["image_filename"] for ann in val_annotations}288 for ann in unified_annotations:289 if ann["image_filename"] not in existing_filenames:290 val_annotations.append(ann)291 existing_filenames.add(ann["image_filename"])292 293 print(f"将valid分割与val分割合并,共 {len(val_annotations)} 条记录")294 unified_annotations = val_annotations295 except Exception as e:296 print(f"合并valid和val分割时出错: {e}")297 298 # 保存统一格式的注释299 if unified_annotations:300 # 对于valid分割,保存为val_unified.json301 save_split = "val" if split == "valid" else split302 print(f"为 {save_split} 创建统一格式注释,共 {len(unified_annotations)} 条记录")303 unified_path = f"annotations/{save_split}_unified.json"304 with open(unified_path, "w", encoding="utf-8") as f:305 json.dump(unified_annotations, f, ensure_ascii=False, indent=2)306 print(f"已保存统一格式注释到: {unified_path}")307 else:308 print(f"警告: {split} 没有有效的注释,跳过创建统一格式文件") 