cy0307/awesome-loop-engineering
Awesome Loop Engineering Dataset A structured dataset of 1025 papers, official docs, tools, benchmarks, patterns, critiques, and implementation guides for recurring AI-agent systems. Resource Atlas · GitHub field guide · Resource selection · Report a correction Dataset Summary Each row connects an original source to its contribution, novelty, impact, publication details, lifecycle stages, audience, evidence type, link status, and… See the full description on the dataset page: https://huggingface.co/datasets/cy0307/awesome-loop-engineering.
35.3k
1#!/usr/bin/env python32"""Check cross-surface counts, release metadata, and generated discovery files."""3 4from __future__ import annotations5 6import csv7import json8import re9import sys10from pathlib import Path11 12from build_hf_card import project_context, render_card13 14 15ROOT = Path(__file__).resolve().parents[1]16REQUIRED_RESOURCE_FIELDS = {17 "row_id",18 "title",19 "url",20 "canonical_url",21 "resource_type",22 "annotation",23 "key_contribution",24 "novelty",25 "impact",26 "signal",27 "signal_strength",28 "collection",29 "user_goal",30 "lifecycle_stages",31 "audience",32 "loop_layer",33 "scope_fit",34 "evidence_class",35 "evidence_tier",36 "source_status",37 "metadata_source",38 "audited_at",39}40ALLOWED_EVIDENCE_TIERS = {"A", "B", "C", "D"}41ALLOWED_SIGNAL_STRENGTHS = {"high", "medium", "contextual", "unverified"}42ALLOWED_SOURCE_STATUSES = {"ok", "local_ok", "restricted", "broken", "unreachable", "local_missing"}43ALLOWED_LOOP_LAYERS = {"model", "agent", "harness", "workflow", "operations", "evaluation", "cross-layer"}44ALLOWED_SCOPE_FITS = {"direct", "enabling", "adjacent"}45 46 47def require(path: Path, snippets: list[str], failures: list[str]) -> None:48 text = path.read_text(encoding="utf-8")49 for snippet in snippets:50 if snippet not in text:51 failures.append(f"{path.relative_to(ROOT)}: missing {snippet!r}")52 53 54def validate_resource_rows(rows: list[dict[str, str]], failures: list[str]) -> None:55 if not rows:56 failures.append("data/resources.csv: no resource rows")57 return58 59 missing_columns = REQUIRED_RESOURCE_FIELDS - set(rows[0])60 if missing_columns:61 failures.append(f"data/resources.csv: missing columns {sorted(missing_columns)}")62 63 seen_urls: set[str] = set()64 seen_ids: set[str] = set()65 for index, row in enumerate(rows, 1):66 row_label = row.get("row_id") or f"row {index}"67 missing = sorted(field for field in REQUIRED_RESOURCE_FIELDS if not row.get(field, "").strip())68 if missing:69 failures.append(f"data/resources.csv: {row_label} missing required values {missing}")70 71 normalized_url = row.get("url", "").strip().lower().rstrip("/")72 if normalized_url in seen_urls:73 failures.append(f"data/resources.csv: duplicate URL at {row_label}: {row.get('url', '')}")74 seen_urls.add(normalized_url)75 76 row_id = row.get("row_id", "")77 if row_id in seen_ids:78 failures.append(f"data/resources.csv: duplicate row_id {row_id}")79 seen_ids.add(row_id)80 expected_id = f"ale-{index:04d}"81 if row_id != expected_id:82 failures.append(f"data/resources.csv: expected {expected_id}, found {row_id or '<blank>'}")83 84 if row.get("evidence_tier") not in ALLOWED_EVIDENCE_TIERS:85 failures.append(f"data/resources.csv: {row_label} has invalid evidence_tier {row.get('evidence_tier')!r}")86 if row.get("signal_strength") not in ALLOWED_SIGNAL_STRENGTHS:87 failures.append(f"data/resources.csv: {row_label} has invalid signal_strength {row.get('signal_strength')!r}")88 if row.get("source_status") not in ALLOWED_SOURCE_STATUSES:89 failures.append(f"data/resources.csv: {row_label} has invalid source_status {row.get('source_status')!r}")90 if row.get("loop_layer") not in ALLOWED_LOOP_LAYERS:91 failures.append(f"data/resources.csv: {row_label} has invalid loop_layer {row.get('loop_layer')!r}")92 if row.get("scope_fit") not in ALLOWED_SCOPE_FITS:93 failures.append(f"data/resources.csv: {row_label} has invalid scope_fit {row.get('scope_fit')!r}")94 95 year = row.get("publication_year", "")96 if year and not re.fullmatch(r"(?:19|20)\d{2}", year):97 failures.append(f"data/resources.csv: {row_label} has invalid publication_year {year!r}")98 if row.get("resource_type") == "Paper":99 for field in ("authors", "publication_year"):100 if not row.get(field, "").strip():101 failures.append(f"data/resources.csv: {row_label} paper is missing {field}")102 if "arxiv.org" in row.get("url", "") and not row.get("arxiv_id", "").strip():103 failures.append(f"data/resources.csv: {row_label} arXiv work is missing arxiv_id")104 105 106def main() -> int:107 context = project_context()108 count = context["RESOURCE_COUNT"]109 version = context["VERSION"]110 source_summary = (111 f"{context['REACHABLE_COUNT']} public links opened successfully, "112 f"{context['RESTRICTED_COUNT']} required access, "113 f"{context['LOCAL_COUNT']} pointed to files in this repository"114 )115 failures: list[str] = []116 117 require(118 ROOT / "README.md",119 [120 f"resources-{count}-",121 f"patterns-{context['PATTERN_COUNT']}-",122 f"contracts-{context['CONTRACT_COUNT']}-",123 f"starters-{context['RUNNABLE_COUNT']}-",124 f"**{count} resources**",125 source_summary,126 f"**{context['RUNNABLE_COUNT']} runtime starters**",127 f"{context['EXECUTABLE_COUNT']} dependency-light executables plus {context['RUNTIME_TEMPLATE_COUNT']} copy/paste runtime templates",128 f"{context['LANGUAGE_COUNT']} language entry points",129 ],130 failures,131 )132 require(133 ROOT / "docs" / "index.html",134 [135 f'content="Explore {count} resources',136 f">{count}</b><span>resources</span>",137 f"Filter {count} resources",138 f"{context['REACHABLE_COUNT']} public links opened successfully, {context['RESTRICTED_COUNT']} required access",139 f'"version": "{version}"',140 ],141 failures,142 )143 require(ROOT / "meta" / "social-preview.html", [f">{count}</b>"], failures)144 distribution_text = (ROOT / "meta" / "DISTRIBUTION.md").read_text(encoding="utf-8")145 share_match = re.search(r"awesome-loop-engineering/(x-v\d+-\d+\.html)", distribution_text)146 if share_match is None:147 failures.append("meta/DISTRIBUTION.md: missing current x share-page link")148 else:149 require(150 ROOT / "docs" / share_match.group(1),151 [f"{count} resources", f"{count} papers"],152 failures,153 )154 require(155 ROOT / "posts" / "launch.md",156 [157 f"# Awesome Loop Engineering v{version}",158 f"{count} resources",159 f"{context['MODEL_COUNT']} model-layer resources",160 f"{context['FIELD_COUNT']}-field CSV",161 f"{context['LANGUAGE_COUNT']} language entry points",162 ],163 failures,164 )165 require(ROOT / "posts" / "launch.zh-CN.md", [f"# Awesome Loop Engineering v{version}", f"{count} 篇论文"], failures)166 require(ROOT / "meta" / "DISTRIBUTION.md", [f"v{version}", f"{count} resources"], failures)167 require(168 ROOT / "design-qa.md",169 [170 f"returns {context['MODEL_COUNT']} of {count} resources ({context['MODEL_PAPER_COUNT']} papers plus Awesome Loop Models)",171 f"{context['REACHABLE_COUNT']} public sources reachable, {context['RESTRICTED_COUNT']} access-restricted, {context['LOCAL_COUNT']} repository-native",172 ],173 failures,174 )175 require(ROOT / "data" / "README.md", [f"Download all {count} resources"], failures)176 177 for translation in sorted(ROOT.glob("README.*.md")):178 require(translation, [count], failures)179 180 with (ROOT / "data" / "resources.csv").open(encoding="utf-8", newline="") as handle:181 resource_rows = list(csv.DictReader(handle))182 csv_count = len(resource_rows)183 if csv_count != int(count):184 failures.append(f"data/resources.csv: expected {count} rows, found {csv_count}")185 validate_resource_rows(resource_rows, failures)186 187 with (ROOT / "data" / "resource_source_audit.csv").open(encoding="utf-8", newline="") as handle:188 audit_rows = list(csv.DictReader(handle))189 if len(audit_rows) != int(count):190 failures.append(f"data/resource_source_audit.csv: expected {count} rows, found {len(audit_rows)}")191 audit_statuses: dict[str, int] = {}192 for row in audit_rows:193 status = row.get("audit_status", "")194 audit_statuses[status] = audit_statuses.get(status, 0) + 1195 expected_statuses = {196 "ok": int(context["REACHABLE_COUNT"]),197 "restricted": int(context["RESTRICTED_COUNT"]),198 "local_ok": int(context["LOCAL_COUNT"]),199 }200 if audit_statuses != expected_statuses:201 failures.append(202 "data/resource_source_audit.csv: status counts do not match the generated resource exports "203 f"({audit_statuses!r} != {expected_statuses!r})"204 )205 206 site_payload = json.loads((ROOT / "docs" / "assets" / "resources.json").read_text(encoding="utf-8"))207 if site_payload.get("count") != int(count) or len(site_payload.get("resources", [])) != int(count):208 failures.append("docs/assets/resources.json: count or resources array is stale")209 210 _, card_failures = render_card()211 failures.extend(card_failures)212 213 if failures:214 print("Project consistency check failed:", file=sys.stderr)215 for failure in failures:216 print(f"- {failure}", file=sys.stderr)217 return 1218 219 print(220 f"Validated {count} resources, {context['MODEL_COUNT']} model-layer resources, "221 f"{context['PATTERN_COUNT']} patterns, {context['CONTRACT_COUNT']} contracts, "222 f"{context['RUNNABLE_COUNT']} runtime starters, and release v{version}."223 )224 return 0225 226 227if __name__ == "__main__":228 raise SystemExit(main())229 