ZZoutian/figure2data-database-v2
figure2data databasev2 — 40K v6 合成科研图表数据集(最终交付版) 生成日期:2026-09-17 生成器:generator 1.6.0 / dataset_generation_revision v6(含两次 hotfix) 规模:40,000 样本(10 图族 / 33 亚型;area、matrix 冻结不生成,set_relation 已删除) 目录结构 databasev2/ ├── figure2data.sqlite3 # 主数据库(2.0 GB:40,000 samples / 280,000 documents) ├── schema/ # sqlite schema ├── shards/ # 数据资产(shard = (样本序号-1)//1000) │ └── shard_000 .. shard_039/ │ ├── images/ # PNG… See the full description on the dataset page: https://huggingface.co/datasets/ZZoutian/figure2data-database-v2.
06.3k
1"""GenerationPlan 测试(方案 §3 test_generation_plan):精确配额 + 确定性。"""2 3from __future__ import annotations4 5import json6from collections import Counter7 8import yaml9 10from databasev1 import constants as C11from databasev1.generation_plan import build_plan, plan_summary, read_plan, write_plan12 13 14def test_plan_exact_quotas():15 """v6 §19:plan 配额精确满足 v6 scoped quota(area frozen / set_relation removed 后重分配)。"""16 from databasev1.generation_plan import scope_quota_v617 18 rows = build_plan()19 n = len(rows)20 assert n == 4000021 22 # v6 硬契约:area/set_relation 均为 023 assert not any(r.chart_family == "area" for r in rows)24 assert not any(r.chart_family == "set_relation" for r in rows)25 26 qcfg = yaml.safe_load(open(C.QUOTA_CONFIG_PATH, encoding="utf-8"))27 scoped = scope_quota_v6(qcfg, n)28 # subtype 精确(对照 scoped quota)29 sub_count = Counter((r.chart_family, r.chart_subtype) for r in rows)30 for fam, spec in scoped["families"].items():31 for sub, cnt in spec["subtypes"].items():32 assert sub_count[(fam, sub)] == int(cnt), \33 f"{fam}/{sub}: {sub_count[(fam, sub)]} != {cnt}"34 assert sum(1 for r in rows if r.chart_family == fam) == int(spec["total"])35 36 # image-level layout 精确 50/25/2537 lay = Counter(r.layout_type for r in rows)38 assert lay["single"] == 2000039 assert lay["subfigure"] == 1000040 assert lay["inset"] == 1000041 42 # difficulty 精确 20/55/2543 diff = Counter(r.difficulty for r in rows)44 assert diff["simple"] == 800045 assert diff["medium"] == 2200046 assert diff["hard"] == 1000047 48 # style group 均匀49 sty = Counter(r.style_group for r in rows)50 assert set(sty.values()) == {8000}, f"style groups not uniform: {sty}"51 52 # sample_id 唯一且连续53 ids = [r.sample_id for r in rows]54 assert len(set(ids)) == 4000055 assert ids[0] == "dbv1_00000001" and ids[-1] == "dbv1_00040000"56 57 # shard 划分58 assert rows[-1].shard_id == 3959 assert all(r.shard_id == i // 1000 for i, r in enumerate(rows))60 61 # recipe_id / parameter_group / template_group 存在62 assert all(r.recipe_id == f"synthetic_{r.chart_subtype}_v1" for r in rows)63 assert all(r.parameter_group for r in rows)64 assert all(r.template_group.startswith("tpl_dbv1_") for r in rows)65 66 67def test_plan_deterministic():68 """两次 build_plan → 完全一致。"""69 a = build_plan()70 b = build_plan()71 assert a == b72 73 74def test_plan_row_schema():75 """plan 行包含 §8 要求的全部字段。"""76 rows = build_plan()77 row = rows[0]78 required = {"sample_id", "global_seed", "sample_seed", "chart_family", "chart_subtype",79 "layout_type", "difficulty", "style_group", "parameter_group",80 "template_group", "recipe_id", "shard_id"}81 d = json.loads(row.to_json())82 assert required <= set(d.keys())83 assert d["global_seed"] == 2026091584 85 86def test_plan_write_read_roundtrip(tmp_path):87 p = tmp_path / "plan.jsonl"88 rows = build_plan()[:50]89 write_plan(rows, p)90 back = read_plan(p)91 assert back == rows92 93 94def test_plan_summary():95 rows = build_plan()96 s = plan_summary(rows)97 assert s["n"] == 4000098 assert s["layouts"] == {"inset": 10000, "single": 20000, "subfigure": 10000}99 assert s["difficulties"] == {"hard": 10000, "medium": 22000, "simple": 8000}100 assert len(s["subtypes"]) == 33 # v6:upset 删除 + area/matrix frozen 不进 plan101 assert s["shards"] == 40102 