cy0307/awesome-loop-engineering
Awesome Loop Engineering Dataset A structured dataset of 1025 papers, official docs, tools, benchmarks, patterns, critiques, and implementation guides for recurring AI-agent systems. Resource Atlas · GitHub field guide · Resource selection · Report a correction Dataset Summary Each row connects an original source to its contribution, novelty, impact, publication details, lifecycle stages, audience, evidence type, link status, and… See the full description on the dataset page: https://huggingface.co/datasets/cy0307/awesome-loop-engineering.
35.3k
1{2 "name": "Evaluation regression",3 "objective": "Detect regressions in agent behavior, connect them to recent prompt, context, or harness changes, and produce a verified repair proposal.",4 "trigger": {5 "type": "scheduled",6 "cadence_or_event": "Nightly after eval completion, and when benchmark scores drop or trace graders fail."7 },8 "intake": {9 "sources": ["eval run results", "failing task IDs", "trace samples", "baseline scores", "recent prompt and harness commits"],10 "selection_rule": "Investigate regression clusters with reproducible evidence; skip runs inside accepted variance or known-flaky sets."11 },12 "workspace": {13 "isolation": "Branch or sandbox with read access to traces and eval artifacts.",14 "allowed_actions": ["targeted eval reruns", "scorer inspection", "small prompt, context, or fixture patches", "report generation"],15 "disallowed_actions": ["scorer changes that hide failures", "benchmark cherry-picking", "broad prompt rewrites", "leaderboard claims"]16 },17 "context": {18 "required_files": ["evaluation rubric", "known-flaky eval list"],19 "runtime_sources": ["baseline traces", "current traces", "model and runtime configuration"]20 },21 "agents": [22 {23 "role": "Investigator",24 "responsibility": "Compare failing traces against passing baseline traces."25 },26 {27 "role": "Hypothesis writer",28 "responsibility": "Classify the likely cause: model behavior, context, tool, scorer, fixture, or harness."29 },30 {31 "role": "Implementer",32 "responsibility": "Patch the smallest plausible cause supported by trace evidence."33 },34 {35 "role": "Verifier",36 "responsibility": "Rerun targeted evals, then a smoke suite, and check for new regressions."37 },38 {39 "role": "Judge",40 "responsibility": "Decide whether evidence supports merging, deferring, or escalating."41 }42 ],43 "verification": {44 "gates": ["targeted failing tasks return to baseline", "no sentinel tasks regress", "trace evidence supports the claimed cause", "score deltas include run IDs and variance caveats"],45 "receipts": ["eval run IDs", "trace excerpts", "hypotheses considered", "rerun scores"]46 },47 "state": {48 "artifacts": ["regression investigation notes", "patch-attempt log"],49 "update_rule": "Record run IDs, failing tasks, hypotheses, patch attempts, rerun scores, and the final decision per regression cluster."50 },51 "budget": {52 "max_retries": 3,53 "max_runtime_minutes": 12054 },55 "escalation": {56 "conditions": ["scorer bug", "benchmark methodology change", "missing private traces", "model-provider incident", "fix risks overfitting"],57 "destination": "Issue or PR tagged for the eval owner with reproducible evidence"58 },59 "exit": {60 "success": "The regression is repaired with verified reruns, or classified as flaky or scorer drift with evidence.",61 "stop_without_success": "Artifacts are missing, retries are exhausted, or the tradeoff requires product judgment."62 }63}64 