OneScience-Group/DMC
DMC Dataset Description DMC is a solvent XTB extxyz dataset for introductory training examples. It contains atomic coordinates, energies, and forces for carbonate molecular systems (VC, EC, PC, DMC, EMC, and DEC) in different configurations, and serves as a standard introductory benchmark dataset for learning and validating machine-learning interatomic potential methods. The training file solvent_xtb_train_200.xyz contains 203 frames (200 training configurations + 3… See the full description on the dataset page: https://huggingface.co/datasets/OneScience-Group/DMC.
028
1#!/usr/bin/env python32"""Validate the standardized DMC extxyz dataset package."""3 4from __future__ import annotations5 6import argparse7import hashlib8import re9from pathlib import Path10 11 12EXPECTED_FILES = {13 "solvent_xtb_train_200.xyz": 203,14 "solvent_xtb_test.xyz": 1000,15}16EXPECTED_ELEMENTS = {"H", "C", "O"}17 18 19def sha256_file(path: Path) -> str:20 h = hashlib.sha256()21 with path.open("rb") as f:22 for chunk in iter(lambda: f.read(1024 * 1024), b""):23 h.update(chunk)24 return h.hexdigest()25 26 27def validate_checksum_manifest(dataset_root: Path, manifest_path: Path) -> None:28 checked = 029 with manifest_path.open("r", encoding="utf-8") as f:30 for line in f:31 line = line.strip()32 if not line or line.startswith("#"):33 continue34 digest, rel = line.split(maxsplit=1)35 path = dataset_root / rel36 if not path.exists():37 raise SystemExit(f"[FAIL] 清单文件不存在: {path}")38 actual = sha256_file(path)39 if actual != digest:40 raise SystemExit(f"[FAIL] SHA256 不一致: {rel}")41 checked += 142 if checked != len(EXPECTED_FILES):43 raise SystemExit(f"[FAIL] SHA256 清单文件数异常: {checked}")44 print(f"[OK] SHA256 清单验证通过: {checked} 个文件")45 46 47def parse_properties(comment: str) -> list[tuple[str, str, int]]:48 match = re.search(r'Properties=("[^"]+"|\S+)', comment)49 if not match:50 raise SystemExit("[FAIL] extxyz comment 缺少 Properties")51 raw = match.group(1).strip('"')52 parts = raw.split(":")53 if len(parts) % 3:54 raise SystemExit(f"[FAIL] Properties 格式异常: {raw}")55 parsed = []56 for i in range(0, len(parts), 3):57 name, kind, width = parts[i], parts[i + 1], int(parts[i + 2])58 parsed.append((name, kind, width))59 return parsed60 61 62def validate_xyz(path: Path, expected_frames: int) -> None:63 frames = 064 elements: set[str] = set()65 min_atoms: int | None = None66 max_atoms = 067 saw_molid = False68 69 with path.open("r", encoding="utf-8") as f:70 while True:71 first = f.readline()72 if not first:73 break74 if not first.strip():75 continue76 try:77 atom_count = int(first.strip())78 except ValueError as exc:79 raise SystemExit(f"[FAIL] 原子数行不是整数: {path}:{frames + 1}") from exc80 comment = f.readline().rstrip("\n")81 if "energy_xtb=" not in comment:82 raise SystemExit(f"[FAIL] comment 缺少 energy_xtb: {path} frame {frames}")83 if "pbc=" not in comment:84 raise SystemExit(f"[FAIL] comment 缺少 pbc: {path} frame {frames}")85 props = parse_properties(comment)86 prop_names = [p[0] for p in props]87 if prop_names[0:2] != ["species", "pos"]:88 raise SystemExit(f"[FAIL] Properties 前两项必须是 species,pos: {path} frame {frames}")89 if "forces_xtb" not in prop_names:90 raise SystemExit(f"[FAIL] Properties 缺少 forces_xtb: {path} frame {frames}")91 saw_molid = saw_molid or ("molID" in prop_names)92 expected_cols = sum(width for _, _, width in props)93 94 for atom_index in range(atom_count):95 atom_line = f.readline()96 if not atom_line:97 raise SystemExit(f"[FAIL] 文件提前结束: {path} frame {frames}")98 cols = atom_line.split()99 if len(cols) != expected_cols:100 raise SystemExit(101 f"[FAIL] 原子列数异常: {path} frame {frames} atom {atom_index}: "102 f"{len(cols)} != {expected_cols}"103 )104 elements.add(cols[0])105 numeric = cols[1:]106 for value in numeric:107 float(value)108 frames += 1109 min_atoms = atom_count if min_atoms is None else min(min_atoms, atom_count)110 max_atoms = max(max_atoms, atom_count)111 112 if frames != expected_frames:113 raise SystemExit(f"[FAIL] 帧数异常: {path.name}: {frames} != {expected_frames}")114 if not elements.issubset(EXPECTED_ELEMENTS):115 raise SystemExit(f"[FAIL] 元素集合异常: {sorted(elements)}")116 if path.name.endswith("test.xyz") and not saw_molid:117 raise SystemExit(f"[FAIL] 测试集未检测到 molID 字段: {path}")118 print(119 f"[OK] {path.name}: frames={frames}, atoms={min_atoms}-{max_atoms}, "120 f"elements={sorted(elements)}"121 )122 123 124def main() -> int:125 parser = argparse.ArgumentParser()126 parser.add_argument("--dataset-root", default="data/DMC")127 parser.add_argument("--checksum-manifest", default="metadata/sha256_manifest.txt")128 args = parser.parse_args()129 130 repo_root = Path.cwd()131 dataset_root = Path(args.dataset_root)132 if not dataset_root.is_absolute():133 dataset_root = repo_root / dataset_root134 checksum_manifest = Path(args.checksum_manifest)135 if not checksum_manifest.is_absolute():136 checksum_manifest = repo_root / checksum_manifest137 138 if not dataset_root.exists():139 raise SystemExit(f"[FAIL] 数据目录不存在: {dataset_root}")140 for name, expected_frames in EXPECTED_FILES.items():141 path = dataset_root / name142 if not path.exists():143 raise SystemExit(f"[FAIL] 缺少文件: {path}")144 if path.stat().st_size <= 0:145 raise SystemExit(f"[FAIL] 文件为空: {path}")146 validate_xyz(path, expected_frames)147 validate_checksum_manifest(dataset_root, checksum_manifest)148 print("[OK] DMC 数据集读取验证通过")149 return 0150 151 152if __name__ == "__main__":153 raise SystemExit(main())154 