Team Ai
Datasetpublic

OneScience-Group/DMC

DMC Dataset Description DMC is a solvent XTB extxyz dataset for introductory training examples. It contains atomic coordinates, energies, and forces for carbonate molecular systems (VC, EC, PC, DMC, EMC, and DEC) in different configurations, and serves as a standard introductory benchmark dataset for learning and validating machine-learning interatomic potential methods. The training file solvent_xtb_train_200.xyz contains 203 frames (200 training configurations + 3… See the full description on the dataset page: https://huggingface.co/datasets/OneScience-Group/DMC.

sourceHugging Facemitupdated 2mo agoView on Hugging Face
0likes28downloads
validate_dmc.py154 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Validate the standardized DMC extxyz dataset package."""3 4from __future__ import annotations5 6import argparse7import hashlib8import re9from pathlib import Path10 11 12EXPECTED_FILES = {13    "solvent_xtb_train_200.xyz": 203,14    "solvent_xtb_test.xyz": 1000,15}16EXPECTED_ELEMENTS = {"H", "C", "O"}17 18 19def sha256_file(path: Path) -> str:20    h = hashlib.sha256()21    with path.open("rb") as f:22        for chunk in iter(lambda: f.read(1024 * 1024), b""):23            h.update(chunk)24    return h.hexdigest()25 26 27def validate_checksum_manifest(dataset_root: Path, manifest_path: Path) -> None:28    checked = 029    with manifest_path.open("r", encoding="utf-8") as f:30        for line in f:31            line = line.strip()32            if not line or line.startswith("#"):33                continue34            digest, rel = line.split(maxsplit=1)35            path = dataset_root / rel36            if not path.exists():37                raise SystemExit(f"[FAIL] 清单文件不存在: {path}")38            actual = sha256_file(path)39            if actual != digest:40                raise SystemExit(f"[FAIL] SHA256 不一致: {rel}")41            checked += 142    if checked != len(EXPECTED_FILES):43        raise SystemExit(f"[FAIL] SHA256 清单文件数异常: {checked}")44    print(f"[OK] SHA256 清单验证通过: {checked} 个文件")45 46 47def parse_properties(comment: str) -> list[tuple[str, str, int]]:48    match = re.search(r'Properties=("[^"]+"|\S+)', comment)49    if not match:50        raise SystemExit("[FAIL] extxyz comment 缺少 Properties")51    raw = match.group(1).strip('"')52    parts = raw.split(":")53    if len(parts) % 3:54        raise SystemExit(f"[FAIL] Properties 格式异常: {raw}")55    parsed = []56    for i in range(0, len(parts), 3):57        name, kind, width = parts[i], parts[i + 1], int(parts[i + 2])58        parsed.append((name, kind, width))59    return parsed60 61 62def validate_xyz(path: Path, expected_frames: int) -> None:63    frames = 064    elements: set[str] = set()65    min_atoms: int | None = None66    max_atoms = 067    saw_molid = False68 69    with path.open("r", encoding="utf-8") as f:70        while True:71            first = f.readline()72            if not first:73                break74            if not first.strip():75                continue76            try:77                atom_count = int(first.strip())78            except ValueError as exc:79                raise SystemExit(f"[FAIL] 原子数行不是整数: {path}:{frames + 1}") from exc80            comment = f.readline().rstrip("\n")81            if "energy_xtb=" not in comment:82                raise SystemExit(f"[FAIL] comment 缺少 energy_xtb: {path} frame {frames}")83            if "pbc=" not in comment:84                raise SystemExit(f"[FAIL] comment 缺少 pbc: {path} frame {frames}")85            props = parse_properties(comment)86            prop_names = [p[0] for p in props]87            if prop_names[0:2] != ["species", "pos"]:88                raise SystemExit(f"[FAIL] Properties 前两项必须是 species,pos: {path} frame {frames}")89            if "forces_xtb" not in prop_names:90                raise SystemExit(f"[FAIL] Properties 缺少 forces_xtb: {path} frame {frames}")91            saw_molid = saw_molid or ("molID" in prop_names)92            expected_cols = sum(width for _, _, width in props)93 94            for atom_index in range(atom_count):95                atom_line = f.readline()96                if not atom_line:97                    raise SystemExit(f"[FAIL] 文件提前结束: {path} frame {frames}")98                cols = atom_line.split()99                if len(cols) != expected_cols:100                    raise SystemExit(101                        f"[FAIL] 原子列数异常: {path} frame {frames} atom {atom_index}: "102                        f"{len(cols)} != {expected_cols}"103                    )104                elements.add(cols[0])105                numeric = cols[1:]106                for value in numeric:107                    float(value)108            frames += 1109            min_atoms = atom_count if min_atoms is None else min(min_atoms, atom_count)110            max_atoms = max(max_atoms, atom_count)111 112    if frames != expected_frames:113        raise SystemExit(f"[FAIL] 帧数异常: {path.name}: {frames} != {expected_frames}")114    if not elements.issubset(EXPECTED_ELEMENTS):115        raise SystemExit(f"[FAIL] 元素集合异常: {sorted(elements)}")116    if path.name.endswith("test.xyz") and not saw_molid:117        raise SystemExit(f"[FAIL] 测试集未检测到 molID 字段: {path}")118    print(119        f"[OK] {path.name}: frames={frames}, atoms={min_atoms}-{max_atoms}, "120        f"elements={sorted(elements)}"121    )122 123 124def main() -> int:125    parser = argparse.ArgumentParser()126    parser.add_argument("--dataset-root", default="data/DMC")127    parser.add_argument("--checksum-manifest", default="metadata/sha256_manifest.txt")128    args = parser.parse_args()129 130    repo_root = Path.cwd()131    dataset_root = Path(args.dataset_root)132    if not dataset_root.is_absolute():133        dataset_root = repo_root / dataset_root134    checksum_manifest = Path(args.checksum_manifest)135    if not checksum_manifest.is_absolute():136        checksum_manifest = repo_root / checksum_manifest137 138    if not dataset_root.exists():139        raise SystemExit(f"[FAIL] 数据目录不存在: {dataset_root}")140    for name, expected_frames in EXPECTED_FILES.items():141        path = dataset_root / name142        if not path.exists():143            raise SystemExit(f"[FAIL] 缺少文件: {path}")144        if path.stat().st_size <= 0:145            raise SystemExit(f"[FAIL] 文件为空: {path}")146        validate_xyz(path, expected_frames)147    validate_checksum_manifest(dataset_root, checksum_manifest)148    print("[OK] DMC 数据集读取验证通过")149    return 0150 151 152if __name__ == "__main__":153    raise SystemExit(main())154