Team Ai
Datasetpublic

cy0307/awesome-loop-engineering

Awesome Loop Engineering Dataset A structured dataset of 1025 papers, official docs, tools, benchmarks, patterns, critiques, and implementation guides for recurring AI-agent systems. Resource Atlas · GitHub field guide · Resource selection · Report a correction   Dataset Summary Each row connects an original source to its contribution, novelty, impact, publication details, lifecycle stages, audience, evidence type, link status, and… See the full description on the dataset page: https://huggingface.co/datasets/cy0307/awesome-loop-engineering.

sourceHugging Facecc0-1.0updated 14d agoView on Hugging Face
3likes5.3kdownloads
check_project_consistency.py229 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""Check cross-surface counts, release metadata, and generated discovery files."""3 4from __future__ import annotations5 6import csv7import json8import re9import sys10from pathlib import Path11 12from build_hf_card import project_context, render_card13 14 15ROOT = Path(__file__).resolve().parents[1]16REQUIRED_RESOURCE_FIELDS = {17    "row_id",18    "title",19    "url",20    "canonical_url",21    "resource_type",22    "annotation",23    "key_contribution",24    "novelty",25    "impact",26    "signal",27    "signal_strength",28    "collection",29    "user_goal",30    "lifecycle_stages",31    "audience",32    "loop_layer",33    "scope_fit",34    "evidence_class",35    "evidence_tier",36    "source_status",37    "metadata_source",38    "audited_at",39}40ALLOWED_EVIDENCE_TIERS = {"A", "B", "C", "D"}41ALLOWED_SIGNAL_STRENGTHS = {"high", "medium", "contextual", "unverified"}42ALLOWED_SOURCE_STATUSES = {"ok", "local_ok", "restricted", "broken", "unreachable", "local_missing"}43ALLOWED_LOOP_LAYERS = {"model", "agent", "harness", "workflow", "operations", "evaluation", "cross-layer"}44ALLOWED_SCOPE_FITS = {"direct", "enabling", "adjacent"}45 46 47def require(path: Path, snippets: list[str], failures: list[str]) -> None:48    text = path.read_text(encoding="utf-8")49    for snippet in snippets:50        if snippet not in text:51            failures.append(f"{path.relative_to(ROOT)}: missing {snippet!r}")52 53 54def validate_resource_rows(rows: list[dict[str, str]], failures: list[str]) -> None:55    if not rows:56        failures.append("data/resources.csv: no resource rows")57        return58 59    missing_columns = REQUIRED_RESOURCE_FIELDS - set(rows[0])60    if missing_columns:61        failures.append(f"data/resources.csv: missing columns {sorted(missing_columns)}")62 63    seen_urls: set[str] = set()64    seen_ids: set[str] = set()65    for index, row in enumerate(rows, 1):66        row_label = row.get("row_id") or f"row {index}"67        missing = sorted(field for field in REQUIRED_RESOURCE_FIELDS if not row.get(field, "").strip())68        if missing:69            failures.append(f"data/resources.csv: {row_label} missing required values {missing}")70 71        normalized_url = row.get("url", "").strip().lower().rstrip("/")72        if normalized_url in seen_urls:73            failures.append(f"data/resources.csv: duplicate URL at {row_label}: {row.get('url', '')}")74        seen_urls.add(normalized_url)75 76        row_id = row.get("row_id", "")77        if row_id in seen_ids:78            failures.append(f"data/resources.csv: duplicate row_id {row_id}")79        seen_ids.add(row_id)80        expected_id = f"ale-{index:04d}"81        if row_id != expected_id:82            failures.append(f"data/resources.csv: expected {expected_id}, found {row_id or '<blank>'}")83 84        if row.get("evidence_tier") not in ALLOWED_EVIDENCE_TIERS:85            failures.append(f"data/resources.csv: {row_label} has invalid evidence_tier {row.get('evidence_tier')!r}")86        if row.get("signal_strength") not in ALLOWED_SIGNAL_STRENGTHS:87            failures.append(f"data/resources.csv: {row_label} has invalid signal_strength {row.get('signal_strength')!r}")88        if row.get("source_status") not in ALLOWED_SOURCE_STATUSES:89            failures.append(f"data/resources.csv: {row_label} has invalid source_status {row.get('source_status')!r}")90        if row.get("loop_layer") not in ALLOWED_LOOP_LAYERS:91            failures.append(f"data/resources.csv: {row_label} has invalid loop_layer {row.get('loop_layer')!r}")92        if row.get("scope_fit") not in ALLOWED_SCOPE_FITS:93            failures.append(f"data/resources.csv: {row_label} has invalid scope_fit {row.get('scope_fit')!r}")94 95        year = row.get("publication_year", "")96        if year and not re.fullmatch(r"(?:19|20)\d{2}", year):97            failures.append(f"data/resources.csv: {row_label} has invalid publication_year {year!r}")98        if row.get("resource_type") == "Paper":99            for field in ("authors", "publication_year"):100                if not row.get(field, "").strip():101                    failures.append(f"data/resources.csv: {row_label} paper is missing {field}")102        if "arxiv.org" in row.get("url", "") and not row.get("arxiv_id", "").strip():103            failures.append(f"data/resources.csv: {row_label} arXiv work is missing arxiv_id")104 105 106def main() -> int:107    context = project_context()108    count = context["RESOURCE_COUNT"]109    version = context["VERSION"]110    source_summary = (111        f"{context['REACHABLE_COUNT']} public links opened successfully, "112        f"{context['RESTRICTED_COUNT']} required access, "113        f"{context['LOCAL_COUNT']} pointed to files in this repository"114    )115    failures: list[str] = []116 117    require(118        ROOT / "README.md",119        [120            f"resources-{count}-",121            f"patterns-{context['PATTERN_COUNT']}-",122            f"contracts-{context['CONTRACT_COUNT']}-",123            f"starters-{context['RUNNABLE_COUNT']}-",124            f"**{count} resources**",125            source_summary,126            f"**{context['RUNNABLE_COUNT']} runtime starters**",127            f"{context['EXECUTABLE_COUNT']} dependency-light executables plus {context['RUNTIME_TEMPLATE_COUNT']} copy/paste runtime templates",128            f"{context['LANGUAGE_COUNT']} language entry points",129        ],130        failures,131    )132    require(133        ROOT / "docs" / "index.html",134        [135            f'content="Explore {count} resources',136            f">{count}</b><span>resources</span>",137            f"Filter {count} resources",138            f"{context['REACHABLE_COUNT']} public links opened successfully, {context['RESTRICTED_COUNT']} required access",139            f'"version": "{version}"',140        ],141        failures,142    )143    require(ROOT / "meta" / "social-preview.html", [f">{count}</b>"], failures)144    distribution_text = (ROOT / "meta" / "DISTRIBUTION.md").read_text(encoding="utf-8")145    share_match = re.search(r"awesome-loop-engineering/(x-v\d+-\d+\.html)", distribution_text)146    if share_match is None:147        failures.append("meta/DISTRIBUTION.md: missing current x share-page link")148    else:149        require(150            ROOT / "docs" / share_match.group(1),151            [f"{count} resources", f"{count} papers"],152            failures,153        )154    require(155        ROOT / "posts" / "launch.md",156        [157            f"# Awesome Loop Engineering v{version}",158            f"{count} resources",159            f"{context['MODEL_COUNT']} model-layer resources",160            f"{context['FIELD_COUNT']}-field CSV",161            f"{context['LANGUAGE_COUNT']} language entry points",162        ],163        failures,164    )165    require(ROOT / "posts" / "launch.zh-CN.md", [f"# Awesome Loop Engineering v{version}", f"{count} 篇论文"], failures)166    require(ROOT / "meta" / "DISTRIBUTION.md", [f"v{version}", f"{count} resources"], failures)167    require(168        ROOT / "design-qa.md",169        [170            f"returns {context['MODEL_COUNT']} of {count} resources ({context['MODEL_PAPER_COUNT']} papers plus Awesome Loop Models)",171            f"{context['REACHABLE_COUNT']} public sources reachable, {context['RESTRICTED_COUNT']} access-restricted, {context['LOCAL_COUNT']} repository-native",172        ],173        failures,174    )175    require(ROOT / "data" / "README.md", [f"Download all {count} resources"], failures)176 177    for translation in sorted(ROOT.glob("README.*.md")):178        require(translation, [count], failures)179 180    with (ROOT / "data" / "resources.csv").open(encoding="utf-8", newline="") as handle:181        resource_rows = list(csv.DictReader(handle))182    csv_count = len(resource_rows)183    if csv_count != int(count):184        failures.append(f"data/resources.csv: expected {count} rows, found {csv_count}")185    validate_resource_rows(resource_rows, failures)186 187    with (ROOT / "data" / "resource_source_audit.csv").open(encoding="utf-8", newline="") as handle:188        audit_rows = list(csv.DictReader(handle))189    if len(audit_rows) != int(count):190        failures.append(f"data/resource_source_audit.csv: expected {count} rows, found {len(audit_rows)}")191    audit_statuses: dict[str, int] = {}192    for row in audit_rows:193        status = row.get("audit_status", "")194        audit_statuses[status] = audit_statuses.get(status, 0) + 1195    expected_statuses = {196        "ok": int(context["REACHABLE_COUNT"]),197        "restricted": int(context["RESTRICTED_COUNT"]),198        "local_ok": int(context["LOCAL_COUNT"]),199    }200    if audit_statuses != expected_statuses:201        failures.append(202            "data/resource_source_audit.csv: status counts do not match the generated resource exports "203            f"({audit_statuses!r} != {expected_statuses!r})"204        )205 206    site_payload = json.loads((ROOT / "docs" / "assets" / "resources.json").read_text(encoding="utf-8"))207    if site_payload.get("count") != int(count) or len(site_payload.get("resources", [])) != int(count):208        failures.append("docs/assets/resources.json: count or resources array is stale")209 210    _, card_failures = render_card()211    failures.extend(card_failures)212 213    if failures:214        print("Project consistency check failed:", file=sys.stderr)215        for failure in failures:216            print(f"- {failure}", file=sys.stderr)217        return 1218 219    print(220        f"Validated {count} resources, {context['MODEL_COUNT']} model-layer resources, "221        f"{context['PATTERN_COUNT']} patterns, {context['CONTRACT_COUNT']} contracts, "222        f"{context['RUNNABLE_COUNT']} runtime starters, and release v{version}."223    )224    return 0225 226 227if __name__ == "__main__":228    raise SystemExit(main())229