Benjamin-eecs/openrsi-commit-runtime-assets
0168
1"""In-container verifier for pr_runtime tasks (graded F2P/P2P reward).2 3This module is the **standalone verifier** that runs inside the task's4Docker container, NOT a helper used at generation time. It is read as5source at generation time, base64-encoded, and embedded into6``tests/test.sh``. At run time the container decodes it back to a file7and invokes it after the test suite has run.8 9Why a graded reward instead of binary pass/fail10------------------------------------------------11SWE-bench resolution is binary: a patch "resolves" the issue iff ALL12FAIL_TO_PASS tests pass AND ALL PASS_TO_PASS tests still pass. That's13the right signal for an eval leaderboard, but a terrible gradient for14RL training — an agent that fixes 4 of 5 failing tests scores the same150.0 as one that fixes nothing.16 17So this verifier emits BOTH:18 * /logs/verifier/reward.txt — the GRADED scalar (training signal,19 which Harbor reads): reward = f2p_rate * p2p_factor20 * /logs/verifier/reward-details.json — carries the strict SWE-bench21 ``resolved`` bool (eval signal) PLUS the full breakdown.22 23Refs: SWE-bench (F2P/P2P semantics), SWE-RL / SWE-Gym (dense reward for24RL), UTBoost (weak-test coverage lets wrong patches pass — hence we25record p2p_count so consumers can judge regression-guard strength).26 27Scoring28-------29Given the baked FAIL_TO_PASS / PASS_TO_PASS test-name lists (computed by30the generation-time two-stage validation) and the agent-run test log:31 32 f2p_rate = (# F2P tests now PASSED) / (# F2P tests)33 p2p_rate = (# P2P tests still PASSED) / (# P2P tests) [1.0 if no P2P]34 p2p_factor = p2p_rate # regressions scale the reward down35 reward = f2p_rate * p2p_factor36 resolved = (all F2P pass) AND (all P2P pass) # strict SWE-bench37 38Oracle invariant: the gold patch flips every F2P and keeps every P2P,39so f2p_rate=1.0, p2p_rate=1.0 -> reward=1.0 and resolved=True. This is40what the T3 oracle gate (reward == 1.0) relies on.41 42Graceful degradation: if the log can't be parsed into per-test43statuses (unrecognized runner output), fall back to the exit-code44reward (1.0 if the suite exited 0, else 0.0) and stamp45``parse_status="fallback_exitcode"`` — never crash, never silently46zero out a real fix.47 48Pure stdlib — uses only ``argparse``, ``json``, ``os``, ``re``, ``sys``.49The 4 per-runner parsers are condensed ports of50``repo2rlenv.log_parsers.*`` kept in lockstep via the unit tests under51``tests/test_pr_runtime_verifier.py``.52"""53 54from __future__ import annotations55 56import argparse57import json58import os59import re60import sys61 62# Canonical statuses. (Plain strings — no typing.Literal so the baked63# module stays import-light when decoded standalone in the container.)64PASSED = "PASSED"65FAILED = "FAILED"66SKIPPED = "SKIPPED"67ERROR = "ERROR"68 69# ---------------------------------------------------------------------------70# Per-runner log parsers (condensed ports of repo2rlenv.log_parsers.*)71# ---------------------------------------------------------------------------72 73_PYTEST_STATUSES = (PASSED, FAILED, SKIPPED, ERROR)74# Keep these patterns in sync with log_parsers/pytest_parser.py. This module75# also runs standalone inside task containers, without a repo2rlenv install.76_PYTEST_PROGRESS = (77 r"\[\s*\d+%\]|\[\s*\d+\s*/\s*\d+\s*\]"78 r"|\d+(?:\.\d+)?(?:us|ms|s)|\d+m \d+s|\d+h \d+m"79)80_PYTEST_VERBOSE_RE = re.compile(81 r"^(?P<name>.+?(?:\[.*?\])?)\s+(?P<status>PASSED|FAILED|SKIPPED|ERROR)"82 rf"(?:\s+\(.*\))?(?:\s+(?:{_PYTEST_PROGRESS}))?$"83)84_PYTEST_SUMMARY_NAME_RE = re.compile(r"^(?P<name>.+?(?:\[.*?\])?)(?: - .*)?$")85# pytest-xdist worker label, e.g. `[gw1] [ 20%] PASSED tests/foo.py::test_x`.86_PYTEST_XDIST_PREFIX_RE = re.compile(rf"^\[gw\d+\]\s+(?:(?:{_PYTEST_PROGRESS})\s+)?")87 88 89def parse_pytest(log: str) -> dict[str, str]:90 """{test_name -> status} from pytest output (verbose or summary)."""91 out: dict[str, str] = {}92 if not log:93 return out94 for raw in log.split("\n"):95 line = raw.strip()96 if not line:97 continue98 line = _PYTEST_XDIST_PREFIX_RE.sub("", line, count=1)99 # Summary lines (STATUS first), preserving the full node ID.100 leading = None101 for st in _PYTEST_STATUSES:102 if line.startswith(st + " ") or line == st:103 leading = st104 break105 if leading is not None:106 work = line[len(leading) :].strip()107 if leading == SKIPPED and re.match(r"^\[\d+\](?:\s|$)", work):108 # Folded skips report a file location, not a parametrized ID.109 tokens = work.split(maxsplit=2)110 if len(tokens) < 2:111 continue112 out[tokens[1]] = leading113 elif m := _PYTEST_SUMMARY_NAME_RE.match(work):114 out[m.group("name")] = leading115 continue116 # Verbose progress (NAME first, STATUS after)117 # Filter non-test output before attempting the backtracking regex.118 head = line.split(None, 1)[0]119 if "::" not in head and not head.endswith(".py"):120 continue121 m = _PYTEST_VERBOSE_RE.match(line)122 if m:123 name = m.group("name")124 if "::" in name or name.endswith(".py"):125 out[name] = m.group("status")126 return out127 128 129_GO_TEST_RE = re.compile(r"^\s*---\s+(?P<status>PASS|FAIL|SKIP):\s+(?P<name>\S+)")130_GO_STATUS = {"PASS": PASSED, "FAIL": FAILED, "SKIP": SKIPPED}131 132 133def parse_go_test(log: str) -> dict[str, str]:134 """{test_name -> status} from `go test -v` output."""135 out: dict[str, str] = {}136 if not log:137 return out138 for raw in log.split("\n"):139 m = _GO_TEST_RE.match(raw)140 if m:141 out[m.group("name")] = _GO_STATUS[m.group("status")]142 return out143 144 145# Keep these patterns in sync with log_parsers/cargo_parser.py.146_CARGO_TEST_RE = re.compile(r"^test\s+(?P<name>\S.*?) \.\.\. (?P<status>ok|FAILED|ignored)\b")147_CARGO_MODE_RE = re.compile(r" - (?:should panic|compile fail|compile)$")148_CARGO_DOCTEST_RE = re.compile(r"^(?P<file>\S+) - (?:(?P<item>.+?) )?\(line \d+\)$")149_CARGO_STATUS = {"ok": PASSED, "FAILED": FAILED, "ignored": SKIPPED}150_CARGO_DOCTEST_RANK = {SKIPPED: 0, PASSED: 1, FAILED: 2}151 152 153def parse_cargo_test(log: str) -> dict[str, str]:154 """{test_name -> status} from `cargo test` output.155 156 Test modes (` - should panic`) are dropped from names, and doctests are157 keyed without their `(line N)` so line shifts don't change identity; the158 worst status among an item's doctests wins.159 """160 out: dict[str, str] = {}161 if not log:162 return out163 for raw in log.split("\n"):164 m = _CARGO_TEST_RE.match(raw)165 if not m:166 continue167 name = _CARGO_MODE_RE.sub("", m.group("name")).rstrip()168 status = _CARGO_STATUS[m.group("status")]169 doctest = _CARGO_DOCTEST_RE.match(name)170 if doctest:171 item = doctest.group("item")172 name = f"{doctest.group('file')} - {item}" if item else doctest.group("file")173 if name in out and _CARGO_DOCTEST_RANK[out[name]] > _CARGO_DOCTEST_RANK[status]:174 continue175 out[name] = status176 return out177 178 179_JEST_FILE_RE = re.compile(r"^ ?(?:PASS|FAIL)\s+(?P<path>\S+\.(?:ts|tsx|js|jsx|mjs|cjs))\b")180_JEST_TEST_RE = re.compile(181 r"^(?P<indent>\s*)(?P<glyph>✓|√|✕|×|✗|○|◯)\s+(?P<name>.+?)(?:\s+\(\d+(?:\.\d+)?\s*m?s\))?$"182)183_JEST_GLYPH = {184 "✓": PASSED,185 "√": PASSED,186 "✕": FAILED,187 "×": FAILED,188 "✗": FAILED,189 "○": SKIPPED,190 "◯": SKIPPED,191}192# Keep these patterns in sync with log_parsers/jest_parser.py.193_ANSI_RE = re.compile(r"\x1b\[[0-9;?]*[A-Za-z]")194_VITEST_MARKER_RE = re.compile(r"^\s*(?:RUN\s+v\d+\.\d+|Test Files\s+\d)", re.MULTILINE)195_VITEST_TEST_RE = re.compile(196 r"^\s*(?P<glyph>[✓×↓□]) (?:\|(?P<project>[^|]+)\| | (?P<label>\S+) )?"197 r"(?P<name>\S+ > .+?)(?P<duration> \d{1,9}ms)?"198 r"(?: \(retry x\d{1,9}\))?(?: \(repeat x\d{1,9}\))?(?: \d{1,9} MB heap used)?(?: \[[^\[\]]*\])?$"199)200_VITEST_STATUS = {"✓": PASSED, "×": FAILED, "↓": SKIPPED, "□": SKIPPED}201 202 203def _parse_vitest(log: str) -> dict[str, str]:204 """{test_name -> status} from vitest `--reporter=verbose` lines only."""205 out: dict[str, str] = {}206 for raw in log.split("\n"):207 m = _VITEST_TEST_RE.match(raw.rstrip())208 if not m:209 continue210 glyph, name = m.group("glyph"), m.group("name")211 if glyph in "↓□" and m.group("duration"):212 name += m.group("duration")213 project = m.group("project") or m.group("label")214 out[f"|{project}| {name}" if project else name] = _VITEST_STATUS[glyph]215 return out216 217 218def parse_jest(log: str) -> dict[str, str]:219 """{test_name -> status} from Jest / Mocha / Vitest output."""220 out: dict[str, str] = {}221 if not log:222 return out223 log = _ANSI_RE.sub("", log)224 if _VITEST_MARKER_RE.search(log):225 return _parse_vitest(log)226 current_file: str | None = None227 describe_stack: list[tuple[int, str]] = []228 last_test_indent: int | None = None229 for raw in log.split("\n"):230 line = raw.rstrip()231 if not line:232 continue233 m = _JEST_FILE_RE.match(line)234 if m:235 current_file = m.group("path")236 describe_stack = []237 last_test_indent = None238 continue239 m = _JEST_TEST_RE.match(line)240 if m:241 indent = len(m.group("indent"))242 name = re.sub(r"^(?:skipped|todo):\s*", "", m.group("name").strip())243 describes = [d for ind, d in describe_stack if ind < indent]244 parts = ([current_file] if current_file else []) + describes + [name]245 out[" > ".join(parts)] = _JEST_GLYPH[m.group("glyph")]246 last_test_indent = indent247 continue248 stripped = line.lstrip()249 if not stripped or stripped.startswith(250 ("Tests:", "Test Suites:", "Snapshots:", "Time:", "Ran all", "●", "→", "✗:")251 ):252 continue253 indent_here = len(line) - len(stripped)254 if current_file and (last_test_indent is None or indent_here < last_test_indent):255 describe_stack = [(i, d) for i, d in describe_stack if i < indent_here]256 describe_stack.append((indent_here, stripped))257 return out258 259 260def _detect_runner(test_cmds: str) -> str:261 joined = test_cmds.lower()262 if "pytest" in joined:263 return "pytest"264 if re.search(r"\bgo\s+test\b", joined):265 return "go"266 if re.search(r"\bcargo\s+test\b", joined):267 return "cargo"268 if any(k in joined for k in ("jest", "mocha", "vitest", "npm test", "yarn test", "pnpm test")):269 return "jest"270 return "unknown"271 272 273def parse_logs(runner: str, log: str) -> dict[str, str]:274 """Dispatch to the right per-runner parser. Empty dict if unknown."""275 if runner == "pytest":276 return parse_pytest(log)277 if runner == "go":278 return parse_go_test(log)279 if runner == "cargo":280 return parse_cargo_test(log)281 if runner == "jest":282 return parse_jest(log)283 return {}284 285 286# ---------------------------------------------------------------------------287# Grading288# ---------------------------------------------------------------------------289def _has_go_parent(name: str, status_map: dict[str, str]) -> bool:290 """Return True if a Go test has an ancestor represented in the status map."""291 parts = name.split("/")292 293 for i in range(1, len(parts)):294 parent = "/".join(parts[:i])295 if status_map.get(parent) == FAILED:296 return True297 298 return False299 300 301def grade(302 fail_to_pass: list[str],303 pass_to_pass: list[str],304 status_map: dict[str, str],305 runner: str | None = None,306) -> dict:307 """Compute the graded reward + strict resolved bool from a status map.308 309 f2p_rate = (# F2P now PASSED) / (# F2P)310 p2p_rate = (# P2P still PASSED) / (# P2P) [1.0 when no P2P]311 reward = f2p_rate * p2p_rate (dense training signal)312 resolved = all F2P pass AND all P2P pass (SWE-bench TRACKED resolution —313 the gold patch always satisfies this, preserving the oracle314 invariant)315 316 Two distinct EVAL signals (see the audit's "tracked vs command resolved"):317 * `resolved` — tracked resolution (above). Gold patch -> True.318 * `command_resolved` — stricter: tracked resolution AND the selected test319 command had NO failures outside F2P/P2P AND exit320 code 0. Computed in main() where the exit code is321 known. A benchmark that wants "the whole command322 passed cleanly" gates on this; SWE-bench-style323 scoring uses `resolved`.324 325 `untracked_failed` are FAILED tests in the run that are neither F2P nor326 P2P (e.g. always-failing/flaky tests pulled in by running a whole test327 file). They don't change the graded `reward` or tracked `resolved`, but328 they block `command_resolved` and are recorded for transparency.329 """330 f2p_total = len(fail_to_pass)331 p2p_total = len(pass_to_pass)332 f2p_set = set(fail_to_pass)333 p2p_set = set(pass_to_pass)334 f2p_passed = sum(1 for t in fail_to_pass if status_map.get(t) == PASSED)335 p2p_passed = sum(1 for t in pass_to_pass if status_map.get(t) == PASSED)336 # Tests that should have stayed green but regressed (PASS->not-pass).337 regressions = [t for t in pass_to_pass if status_map.get(t) != PASSED]338 339 # FAILED tests outside the tracked sets — the selected command isn't clean.340 if runner == "go":341 failed_tests = {342 t343 for t, s in status_map.items()344 if s == FAILED and t not in f2p_set and t not in p2p_set345 }346 347 untracked_failed = sorted(t for t in failed_tests if not _has_go_parent(t, status_map))348 else:349 untracked_failed = sorted(350 t351 for t, s in status_map.items()352 if s == FAILED and t not in f2p_set and t not in p2p_set353 )354 355 f2p_rate = (f2p_passed / f2p_total) if f2p_total else 0.0356 p2p_rate = (p2p_passed / p2p_total) if p2p_total else 1.0357 reward = f2p_rate * p2p_rate358 resolved = f2p_total > 0 and f2p_passed == f2p_total and p2p_passed == p2p_total359 360 return {361 "reward": round(max(0.0, min(1.0, reward)), 6),362 "resolved": resolved,363 "f2p_total": f2p_total,364 "f2p_passed": f2p_passed,365 "f2p_rate": round(f2p_rate, 6),366 "p2p_total": p2p_total,367 "p2p_passed": p2p_passed,368 "p2p_rate": round(p2p_rate, 6),369 "regressions": sorted(regressions),370 "untracked_failed_count": len(untracked_failed),371 "untracked_failed": untracked_failed[:20], # cap the list372 }373 374 375# ---------------------------------------------------------------------------376# Entry point (invoked by tests/test.sh inside the container)377# ---------------------------------------------------------------------------378 379 380def _read_json_list(path: str) -> list[str]:381 try:382 with open(path, encoding="utf-8") as f:383 data = json.load(f)384 return [str(x) for x in data] if isinstance(data, list) else []385 except (OSError, ValueError):386 return []387 388 389def _read_text(path: str) -> str:390 try:391 with open(path, encoding="utf-8") as f:392 return f.read()393 except (OSError, UnicodeDecodeError):394 return ""395 396 397def main(argv: list[str] | None = None) -> int:398 p = argparse.ArgumentParser(description="pr_runtime graded F2P/P2P verifier")399 p.add_argument("--log", required=True, help="captured test-run log file")400 p.add_argument("--f2p", required=True, help="JSON file: FAIL_TO_PASS test names")401 p.add_argument("--p2p", required=True, help="JSON file: PASS_TO_PASS test names")402 p.add_argument("--runner", default="", help="pytest|go|cargo|jest (else auto-detect)")403 p.add_argument("--test-cmds", default="", help="test command string (runner auto-detect)")404 p.add_argument("--exit-code", type=int, default=1, help="test suite exit code (fallback)")405 p.add_argument("--out-dir", default="/logs/verifier", help="where to write reward.{txt,json}")406 args = p.parse_args(argv)407 408 log = _read_text(args.log)409 f2p = _read_json_list(args.f2p)410 p2p = _read_json_list(args.p2p)411 runner = args.runner.strip() or _detect_runner(args.test_cmds)412 413 status_map = parse_logs(runner, log)414 415 if not status_map:416 # Unparseable runner output → fall back to the binary exit-code reward417 # (a coarse TRAINING signal) so we never silently zero a real fix on an418 # unrecognized format. But `resolved` is the strict EVAL signal: without419 # parsed per-test status we have NO evidence the declared FAIL_TO_PASS420 # tests passed, so when an F2P oracle exists we must NOT claim resolved.421 # (resolved stays exit-code-based only when there's no declared oracle,422 # e.g. --skip-validation.)423 reward = 1.0 if args.exit_code == 0 else 0.0424 has_oracle = len(f2p) > 0425 resolved = (args.exit_code == 0) and not has_oracle426 breakdown = {427 "reward": reward,428 "resolved": resolved,429 "command_resolved": bool(resolved and args.exit_code == 0),430 "parse_status": "fallback_exitcode",431 "eval_trustworthy": not has_oracle,432 "runner": runner,433 "f2p_total": len(f2p),434 "p2p_total": len(p2p),435 "exit_code": args.exit_code,436 }437 else:438 breakdown = grade(f2p, p2p, status_map, runner)439 breakdown["parse_status"] = "ok"440 breakdown["runner"] = runner441 breakdown["tests_parsed"] = len(status_map)442 breakdown["exit_code"] = args.exit_code443 # Strict eval signal: tracked resolution AND a clean command (no444 # untracked failures, exit code 0). Benchmarks wanting "the whole test445 # command passed" gate on this; SWE-bench-style scoring uses `resolved`.446 breakdown["command_resolved"] = bool(447 breakdown["resolved"]448 and breakdown["untracked_failed_count"] == 0449 and args.exit_code == 0450 )451 reward = breakdown["reward"]452 453 os.makedirs(args.out_dir, exist_ok=True)454 with open(os.path.join(args.out_dir, "reward.txt"), "w", encoding="utf-8") as f:455 f.write(f"{reward:.6f}\n")456 with open(os.path.join(args.out_dir, "reward-details.json"), "w", encoding="utf-8") as f:457 json.dump(breakdown, f, indent=2)458 459 print(json.dumps(breakdown, indent=2))460 return 0461 462 463if __name__ == "__main__":464 sys.exit(main())465 466 467__all__ = [468 "grade",469 "main",470 "parse_cargo_test",471 "parse_go_test",472 "parse_jest",473 "parse_logs",474 "parse_pytest",475]476 