Team Ai
Datasetpublic

ZZoutian/figure2data-database-v2

figure2data databasev2 — 40K v6 合成科研图表数据集(最终交付版) 生成日期:2026-09-17 生成器:generator 1.6.0 / dataset_generation_revision v6(含两次 hotfix) 规模:40,000 样本(10 图族 / 33 亚型;area、matrix 冻结不生成,set_relation 已删除) 目录结构 databasev2/ ├── figure2data.sqlite3 # 主数据库(2.0 GB:40,000 samples / 280,000 documents) ├── schema/ # sqlite schema ├── shards/ # 数据资产(shard = (样本序号-1)//1000) │ └── shard_000 .. shard_039/ │ ├── images/ # PNG… See the full description on the dataset page: https://huggingface.co/datasets/ZZoutian/figure2data-database-v2.

sourceHugging Faceupdated 24d agoView on Hugging Face
0likes6.3kdownloads
test_grouped_split.py60 linesDownload Raw Back to tests
1"""§52.9 Grouped Split Test — parameter_group 原子切分。"""2 3from __future__ import annotations4 5import random6 7from databasev1.split.grouped_split import assert_no_leakage, grouped_split8 9 10def _records(n_groups: int = 40, per_group: int = 5, seed: int = 7):11    rng = random.Random(seed)12    recs = []13    for g in range(n_groups):14        for i in range(rng.randint(1, per_group)):15            recs.append({"sample_id": f"s_{g:03d}_{i:02d}", "parameter_group": f"pg_{g:03d}"})16    return recs17 18 19def test_same_group_never_spans_splits():20    recs = _records()21    out = grouped_split(recs)22    split_of = {}23    for split, rs in out.items():24        for r in rs:25            split_of[r["sample_id"]] = split26    # 原子性27    assert_no_leakage(recs, split_of)28 29 30def test_all_records_assigned_exactly_once():31    recs = _records()32    out = grouped_split(recs)33    assigned = [r["sample_id"] for rs in out.values() for r in rs]34    assert sorted(assigned) == sorted(r["sample_id"] for r in recs)35    assert len(assigned) == len(set(assigned))36 37 38def test_ratios_approximate_70_15_15():39    recs = _records(n_groups=200, per_group=8)40    out = grouped_split(recs)41    n = len(recs)42    assert abs(len(out["train"]) / n - 0.70) < 0.0543    assert abs(len(out["validation"]) / n - 0.15) < 0.0544    assert abs(len(out["test"]) / n - 0.15) < 0.0545 46 47def test_deterministic():48    recs = _records()49    a = grouped_split(recs)50    b = grouped_split(recs)51    for k in a:52        assert [r["sample_id"] for r in a[k]] == [r["sample_id"] for r in b[k]]53 54 55def test_single_group_all_in_one_split():56    recs = [{"sample_id": f"s{i}", "parameter_group": "only"} for i in range(10)]57    out = grouped_split(recs)58    splits_with_records = [k for k, v in out.items() if v]59    assert len(splits_with_records) == 160 
ZZoutian/figure2data-database-v2 · Team Ai