Team Ai
Datasetpublic

OneScience-Group/cfd_benchmark

CFD Benchmark Dataset Description CFD Benchmark is a comprehensive benchmark dataset for training and evaluating neural PDE solvers. It contains six standard tasks on regular grids, structured grids, and irregular geometries. The dataset was used for unified evaluation in the ICML 2024 paper Transolver, and some of its data originates from the FNO and Geo-FNO works. The dataset contains six subsets: airfoil, darcy, elasticity, ns, pipe, and plas. Paper: Transolver:… See the full description on the dataset page: https://huggingface.co/datasets/OneScience-Group/cfd_benchmark.

sourceHugging Faceotherupdated 2mo agoView on Hugging Face
1likes356downloads
validate_dataset.py127 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Validate CFD_Benchmark dataset files without loading full arrays into memory."""3 4from __future__ import annotations5 6import argparse7import csv8import hashlib9import json10from pathlib import Path11from typing import Any12 13 14def sha256_file(path: Path, chunk_size: int = 1024 * 1024 * 16) -> str:15    digest = hashlib.sha256()16    with path.open("rb") as handle:17        for chunk in iter(lambda: handle.read(chunk_size), b""):18            digest.update(chunk)19    return digest.hexdigest()20 21 22def inspect_npy(path: Path) -> dict[str, Any]:23    import numpy as np24 25    array = np.load(path, mmap_mode="r")26    return {"format": "npy", "shape": list(array.shape), "dtype": str(array.dtype)}27 28 29def inspect_mat(path: Path) -> dict[str, Any]:30    info: dict[str, Any] = {"format": "matlab_v5", "variables": []}31    try:32        import scipy.io as scio33 34        variables = []35        for name, shape, dtype in scio.whosmat(path):36            variables.append({"name": name, "shape": list(shape), "dtype": str(dtype)})37        info["variables"] = variables38        info["metadata_mode"] = "scipy.io.whosmat"39    except Exception as exc:  # scipy is optional for integrity-only checks.40        info["metadata_mode"] = "header_only"41        info["metadata_warning"] = f"scipy.io.whosmat unavailable: {type(exc).__name__}: {exc}"42        with path.open("rb") as handle:43            info["header_prefix"] = handle.read(64).decode("latin1", errors="replace")44    return info45 46 47def load_manifest(path: Path) -> list[dict[str, str]]:48    with path.open("r", newline="") as handle:49        return list(csv.DictReader(handle))50 51 52def main() -> int:53    parser = argparse.ArgumentParser()54    parser.add_argument("--data-root", default="data", help="Dataset data root")55    parser.add_argument(56        "--file-manifest",57        default="metadata/file_manifest.csv",58        help="CSV with rel_path,size,sha256 entries",59    )60    parser.add_argument("--skip-sha256", action="store_true", help="Only check names and sizes")61    parser.add_argument("--report", default="reports/dataset_validation.json")62    args = parser.parse_args()63 64    data_root = Path(args.data_root)65    manifest_path = Path(args.file_manifest)66    report_path = Path(args.report)67    report_path.parent.mkdir(parents=True, exist_ok=True)68 69    rows = load_manifest(manifest_path)70    results = []71    ok = True72    for row in rows:73        rel_path = row["rel_path"]74        expected_size = int(row["size"])75        expected_sha256 = row["sha256"]76        path = data_root / rel_path77        item: dict[str, Any] = {78            "rel_path": rel_path,79            "exists": path.exists(),80            "expected_size": expected_size,81            "expected_sha256": expected_sha256,82        }83        if not path.exists():84            item["status"] = "missing"85            ok = False86            results.append(item)87            continue88 89        actual_size = path.stat().st_size90        item["actual_size"] = actual_size91        if actual_size != expected_size:92            item["status"] = "size_mismatch"93            ok = False94        else:95            item["status"] = "ok"96 97        if not args.skip_sha256:98            actual_sha256 = sha256_file(path)99            item["actual_sha256"] = actual_sha256100            if actual_sha256 != expected_sha256:101                item["status"] = "sha256_mismatch"102                ok = False103 104        if path.suffix == ".npy":105            item["schema"] = inspect_npy(path)106        elif path.suffix == ".mat":107            item["schema"] = inspect_mat(path)108        results.append(item)109 110    report = {111        "ok": ok,112        "data_root": str(data_root),113        "file_manifest": str(manifest_path),114        "checked_files": len(results),115        "results": results,116    }117    report_path.write_text(json.dumps(report, indent=2, ensure_ascii=False) + "\n")118    if ok:119        print(f"DATASET_VALIDATION_OK checked_files={len(results)} report={report_path}")120        return 0121    print(f"DATASET_VALIDATION_FAILED report={report_path}")122    return 1123 124 125if __name__ == "__main__":126    raise SystemExit(main())127