Team Ai
Datasetpublic

OneScience-Group/cfdbench

CFDBench Dataset Description CFDBench is a large-scale benchmark dataset for machine learning methods in computational fluid dynamics, designed to evaluate the generalization capabilities of neural operators under unseen boundary conditions, fluid properties, and geometries. The dataset contains four classic CFD problems: lid-driven cavity flow (cavity), laminar pipe flow (tube), step dam-break flow (dam), and flow around a cylinder (cylinder). For each problem… See the full description on the dataset page: https://huggingface.co/datasets/OneScience-Group/cfdbench.

sourceHugging Faceapache-2.0updated 2mo agoView on Hugging Face
0likes105downloads
validate_cfdbench_dataset.py137 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Validate the standardized CFDBench dataset package."""3 4from __future__ import annotations5 6import argparse7import hashlib8import json9import sys10from collections import defaultdict11from pathlib import Path12 13import numpy as np14 15 16REPO_ROOT = Path(__file__).resolve().parents[1]17DATA_ROOT = REPO_ROOT / "data"18CHECKSUM_PATH = REPO_ROOT / "files_sha256.jsonl"19PROBLEMS = ("cavity", "tube", "dam", "cylinder")20SUBSETS = ("bc", "geo", "prop")21REQUIRED_CASE_FILES = ("case.json", "u.npy", "v.npy")22OPTIONAL_EXTENSIONS = {".png"}23 24 25def fail(message: str) -> None:26    print(f"[FAIL] {message}")27    raise SystemExit(1)28 29 30def ok(message: str) -> None:31    print(f"[OK] {message}")32 33 34def warn(message: str) -> None:35    print(f"[WARN] {message}")36 37 38def sha256_file(path: Path) -> str:39    digest = hashlib.sha256()40    with path.open("rb") as handle:41        for chunk in iter(lambda: handle.read(1024 * 1024), b""):42            digest.update(chunk)43    return digest.hexdigest()44 45 46def iter_case_dirs() -> list[Path]:47    case_dirs = []48    for problem in PROBLEMS:49        problem_dir = DATA_ROOT / problem50        if not problem_dir.is_dir():51            fail(f"missing problem directory: {problem_dir}")52        for subset in SUBSETS:53            subset_dir = problem_dir / subset54            if subset_dir.is_dir():55                cases = sorted(subset_dir.glob("case*"), key=lambda p: int(p.name[4:]))56                if not cases:57                    fail(f"empty subset directory: {subset_dir}")58                case_dirs.extend(cases)59    return case_dirs60 61 62def validate_structure(sample_cases_per_subset: int) -> None:63    if not DATA_ROOT.is_dir():64        fail(f"dataset data root does not exist: {DATA_ROOT}")65    counts = defaultdict(int)66    sampled = 067    for problem in PROBLEMS:68        for subset in SUBSETS:69            subset_dir = DATA_ROOT / problem / subset70            if not subset_dir.is_dir():71                continue72            cases = sorted(subset_dir.glob("case*"), key=lambda p: int(p.name[4:]))73            counts[f"{problem}/{subset}"] = len(cases)74            for case_dir in cases[:sample_cases_per_subset]:75                for name in REQUIRED_CASE_FILES:76                    if not (case_dir / name).is_file():77                        fail(f"missing required case file: {case_dir / name}")78                params = json.loads((case_dir / "case.json").read_text(encoding="utf-8"))79                if not isinstance(params, dict) or not params:80                    fail(f"case.json must be a non-empty object: {case_dir / 'case.json'}")81                u = np.load(case_dir / "u.npy", mmap_mode="r")82                v = np.load(case_dir / "v.npy", mmap_mode="r")83                if u.shape != v.shape:84                    fail(f"u/v shape mismatch: {case_dir}: {u.shape} vs {v.shape}")85                if u.ndim != 3:86                    fail(f"u/v arrays must be 3D time/grid arrays: {case_dir}: {u.shape}")87                if not np.issubdtype(u.dtype, np.number) or not np.issubdtype(v.dtype, np.number):88                    fail(f"u/v dtype must be numeric: {case_dir}: {u.dtype}, {v.dtype}")89                if not np.isfinite(np.asarray(u[0])).all() or not np.isfinite(np.asarray(v[0])).all():90                    fail(f"first frame contains non-finite values: {case_dir}")91                sampled += 192    if not counts:93        fail("no CFDBench subsets found")94    ok(f"dataset structure and sampled arrays are valid: {sum(counts.values())} cases, sampled {sampled}")95 96 97def verify_checksums(full_hash: bool) -> None:98    if not CHECKSUM_PATH.exists():99        warn(f"checksum manifest is not present: {CHECKSUM_PATH}")100        return101    records = []102    for line in CHECKSUM_PATH.read_text(encoding="utf-8").splitlines():103        if line.strip():104            records.append(json.loads(line))105    if not records:106        fail("checksum manifest is empty")107    for record in records:108        path = REPO_ROOT / record["path"]109        if not path.is_file():110            fail(f"checksum entry points to missing file: {path}")111        size = path.stat().st_size112        if size != record["size"]:113            fail(f"size mismatch for {path}: expected {record['size']}, got {size}")114        if full_hash:115            digest = sha256_file(path)116            if digest != record["sha256"]:117                fail(f"sha256 mismatch for {path}")118    mode = "size+sha256" if full_hash else "size"119    ok(f"checksum manifest verified in {mode} mode: {len(records)} files")120 121 122def main() -> int:123    parser = argparse.ArgumentParser()124    parser.add_argument("--sample-cases-per-subset", type=int, default=2)125    parser.add_argument("--full-hash", action="store_true")126    args = parser.parse_args()127    if args.sample_cases_per_subset < 1:128        fail("--sample-cases-per-subset must be positive")129    validate_structure(args.sample_cases_per_subset)130    verify_checksums(args.full_hash)131    ok("dataset validation completed")132    return 0133 134 135if __name__ == "__main__":136    sys.exit(main())137