Team Ai
Datasetpublic

Benjamin-eecs/openrsi-commit-runtime-assets

sourceHugging Faceupdated 16d agoView on Hugging Face
0likes168downloads
verifier.py476 linesDownload Raw Back to tests
1"""In-container verifier for pr_runtime tasks (graded F2P/P2P reward).2 3This module is the **standalone verifier** that runs inside the task's4Docker container, NOT a helper used at generation time. It is read as5source at generation time, base64-encoded, and embedded into6``tests/test.sh``. At run time the container decodes it back to a file7and invokes it after the test suite has run.8 9Why a graded reward instead of binary pass/fail10------------------------------------------------11SWE-bench resolution is binary: a patch "resolves" the issue iff ALL12FAIL_TO_PASS tests pass AND ALL PASS_TO_PASS tests still pass. That's13the right signal for an eval leaderboard, but a terrible gradient for14RL training — an agent that fixes 4 of 5 failing tests scores the same150.0 as one that fixes nothing.16 17So this verifier emits BOTH:18  * /logs/verifier/reward.txt  — the GRADED scalar (training signal,19        which Harbor reads):  reward = f2p_rate * p2p_factor20  * /logs/verifier/reward-details.json — carries the strict SWE-bench21        ``resolved`` bool (eval signal) PLUS the full breakdown.22 23Refs: SWE-bench (F2P/P2P semantics), SWE-RL / SWE-Gym (dense reward for24RL), UTBoost (weak-test coverage lets wrong patches pass — hence we25record p2p_count so consumers can judge regression-guard strength).26 27Scoring28-------29Given the baked FAIL_TO_PASS / PASS_TO_PASS test-name lists (computed by30the generation-time two-stage validation) and the agent-run test log:31 32  f2p_rate   = (# F2P tests now PASSED) / (# F2P tests)33  p2p_rate   = (# P2P tests still PASSED) / (# P2P tests)   [1.0 if no P2P]34  p2p_factor = p2p_rate          # regressions scale the reward down35  reward     = f2p_rate * p2p_factor36  resolved   = (all F2P pass) AND (all P2P pass)            # strict SWE-bench37 38Oracle invariant: the gold patch flips every F2P and keeps every P2P,39so f2p_rate=1.0, p2p_rate=1.0 -> reward=1.0 and resolved=True. This is40what the T3 oracle gate (reward == 1.0) relies on.41 42Graceful degradation: if the log can't be parsed into per-test43statuses (unrecognized runner output), fall back to the exit-code44reward (1.0 if the suite exited 0, else 0.0) and stamp45``parse_status="fallback_exitcode"`` — never crash, never silently46zero out a real fix.47 48Pure stdlib — uses only ``argparse``, ``json``, ``os``, ``re``, ``sys``.49The 4 per-runner parsers are condensed ports of50``repo2rlenv.log_parsers.*`` kept in lockstep via the unit tests under51``tests/test_pr_runtime_verifier.py``.52"""53 54from __future__ import annotations55 56import argparse57import json58import os59import re60import sys61 62# Canonical statuses. (Plain strings — no typing.Literal so the baked63# module stays import-light when decoded standalone in the container.)64PASSED = "PASSED"65FAILED = "FAILED"66SKIPPED = "SKIPPED"67ERROR = "ERROR"68 69# ---------------------------------------------------------------------------70# Per-runner log parsers (condensed ports of repo2rlenv.log_parsers.*)71# ---------------------------------------------------------------------------72 73_PYTEST_STATUSES = (PASSED, FAILED, SKIPPED, ERROR)74# Keep these patterns in sync with log_parsers/pytest_parser.py. This module75# also runs standalone inside task containers, without a repo2rlenv install.76_PYTEST_PROGRESS = (77    r"\[\s*\d+%\]|\[\s*\d+\s*/\s*\d+\s*\]"78    r"|\d+(?:\.\d+)?(?:us|ms|s)|\d+m \d+s|\d+h \d+m"79)80_PYTEST_VERBOSE_RE = re.compile(81    r"^(?P<name>.+?(?:\[.*?\])?)\s+(?P<status>PASSED|FAILED|SKIPPED|ERROR)"82    rf"(?:\s+\(.*\))?(?:\s+(?:{_PYTEST_PROGRESS}))?$"83)84_PYTEST_SUMMARY_NAME_RE = re.compile(r"^(?P<name>.+?(?:\[.*?\])?)(?: - .*)?$")85# pytest-xdist worker label, e.g. `[gw1] [ 20%] PASSED tests/foo.py::test_x`.86_PYTEST_XDIST_PREFIX_RE = re.compile(rf"^\[gw\d+\]\s+(?:(?:{_PYTEST_PROGRESS})\s+)?")87 88 89def parse_pytest(log: str) -> dict[str, str]:90    """{test_name -> status} from pytest output (verbose or summary)."""91    out: dict[str, str] = {}92    if not log:93        return out94    for raw in log.split("\n"):95        line = raw.strip()96        if not line:97            continue98        line = _PYTEST_XDIST_PREFIX_RE.sub("", line, count=1)99        # Summary lines (STATUS first), preserving the full node ID.100        leading = None101        for st in _PYTEST_STATUSES:102            if line.startswith(st + " ") or line == st:103                leading = st104                break105        if leading is not None:106            work = line[len(leading) :].strip()107            if leading == SKIPPED and re.match(r"^\[\d+\](?:\s|$)", work):108                # Folded skips report a file location, not a parametrized ID.109                tokens = work.split(maxsplit=2)110                if len(tokens) < 2:111                    continue112                out[tokens[1]] = leading113            elif m := _PYTEST_SUMMARY_NAME_RE.match(work):114                out[m.group("name")] = leading115            continue116        # Verbose progress (NAME first, STATUS after)117        # Filter non-test output before attempting the backtracking regex.118        head = line.split(None, 1)[0]119        if "::" not in head and not head.endswith(".py"):120            continue121        m = _PYTEST_VERBOSE_RE.match(line)122        if m:123            name = m.group("name")124            if "::" in name or name.endswith(".py"):125                out[name] = m.group("status")126    return out127 128 129_GO_TEST_RE = re.compile(r"^\s*---\s+(?P<status>PASS|FAIL|SKIP):\s+(?P<name>\S+)")130_GO_STATUS = {"PASS": PASSED, "FAIL": FAILED, "SKIP": SKIPPED}131 132 133def parse_go_test(log: str) -> dict[str, str]:134    """{test_name -> status} from `go test -v` output."""135    out: dict[str, str] = {}136    if not log:137        return out138    for raw in log.split("\n"):139        m = _GO_TEST_RE.match(raw)140        if m:141            out[m.group("name")] = _GO_STATUS[m.group("status")]142    return out143 144 145# Keep these patterns in sync with log_parsers/cargo_parser.py.146_CARGO_TEST_RE = re.compile(r"^test\s+(?P<name>\S.*?) \.\.\. (?P<status>ok|FAILED|ignored)\b")147_CARGO_MODE_RE = re.compile(r" - (?:should panic|compile fail|compile)$")148_CARGO_DOCTEST_RE = re.compile(r"^(?P<file>\S+) - (?:(?P<item>.+?) )?\(line \d+\)$")149_CARGO_STATUS = {"ok": PASSED, "FAILED": FAILED, "ignored": SKIPPED}150_CARGO_DOCTEST_RANK = {SKIPPED: 0, PASSED: 1, FAILED: 2}151 152 153def parse_cargo_test(log: str) -> dict[str, str]:154    """{test_name -> status} from `cargo test` output.155 156    Test modes (` - should panic`) are dropped from names, and doctests are157    keyed without their `(line N)` so line shifts don't change identity; the158    worst status among an item's doctests wins.159    """160    out: dict[str, str] = {}161    if not log:162        return out163    for raw in log.split("\n"):164        m = _CARGO_TEST_RE.match(raw)165        if not m:166            continue167        name = _CARGO_MODE_RE.sub("", m.group("name")).rstrip()168        status = _CARGO_STATUS[m.group("status")]169        doctest = _CARGO_DOCTEST_RE.match(name)170        if doctest:171            item = doctest.group("item")172            name = f"{doctest.group('file')} - {item}" if item else doctest.group("file")173            if name in out and _CARGO_DOCTEST_RANK[out[name]] > _CARGO_DOCTEST_RANK[status]:174                continue175        out[name] = status176    return out177 178 179_JEST_FILE_RE = re.compile(r"^ ?(?:PASS|FAIL)\s+(?P<path>\S+\.(?:ts|tsx|js|jsx|mjs|cjs))\b")180_JEST_TEST_RE = re.compile(181    r"^(?P<indent>\s*)(?P<glyph>✓|√|✕|×|✗|○|◯)\s+(?P<name>.+?)(?:\s+\(\d+(?:\.\d+)?\s*m?s\))?$"182)183_JEST_GLYPH = {184    "✓": PASSED,185    "√": PASSED,186    "✕": FAILED,187    "×": FAILED,188    "✗": FAILED,189    "○": SKIPPED,190    "◯": SKIPPED,191}192# Keep these patterns in sync with log_parsers/jest_parser.py.193_ANSI_RE = re.compile(r"\x1b\[[0-9;?]*[A-Za-z]")194_VITEST_MARKER_RE = re.compile(r"^\s*(?:RUN\s+v\d+\.\d+|Test Files\s+\d)", re.MULTILINE)195_VITEST_TEST_RE = re.compile(196    r"^\s*(?P<glyph>[✓×↓□]) (?:\|(?P<project>[^|]+)\| | (?P<label>\S+)  )?"197    r"(?P<name>\S+ > .+?)(?P<duration> \d{1,9}ms)?"198    r"(?: \(retry x\d{1,9}\))?(?: \(repeat x\d{1,9}\))?(?: \d{1,9} MB heap used)?(?: \[[^\[\]]*\])?$"199)200_VITEST_STATUS = {"✓": PASSED, "×": FAILED, "↓": SKIPPED, "□": SKIPPED}201 202 203def _parse_vitest(log: str) -> dict[str, str]:204    """{test_name -> status} from vitest `--reporter=verbose` lines only."""205    out: dict[str, str] = {}206    for raw in log.split("\n"):207        m = _VITEST_TEST_RE.match(raw.rstrip())208        if not m:209            continue210        glyph, name = m.group("glyph"), m.group("name")211        if glyph in "↓□" and m.group("duration"):212            name += m.group("duration")213        project = m.group("project") or m.group("label")214        out[f"|{project}| {name}" if project else name] = _VITEST_STATUS[glyph]215    return out216 217 218def parse_jest(log: str) -> dict[str, str]:219    """{test_name -> status} from Jest / Mocha / Vitest output."""220    out: dict[str, str] = {}221    if not log:222        return out223    log = _ANSI_RE.sub("", log)224    if _VITEST_MARKER_RE.search(log):225        return _parse_vitest(log)226    current_file: str | None = None227    describe_stack: list[tuple[int, str]] = []228    last_test_indent: int | None = None229    for raw in log.split("\n"):230        line = raw.rstrip()231        if not line:232            continue233        m = _JEST_FILE_RE.match(line)234        if m:235            current_file = m.group("path")236            describe_stack = []237            last_test_indent = None238            continue239        m = _JEST_TEST_RE.match(line)240        if m:241            indent = len(m.group("indent"))242            name = re.sub(r"^(?:skipped|todo):\s*", "", m.group("name").strip())243            describes = [d for ind, d in describe_stack if ind < indent]244            parts = ([current_file] if current_file else []) + describes + [name]245            out[" > ".join(parts)] = _JEST_GLYPH[m.group("glyph")]246            last_test_indent = indent247            continue248        stripped = line.lstrip()249        if not stripped or stripped.startswith(250            ("Tests:", "Test Suites:", "Snapshots:", "Time:", "Ran all", "●", "→", "✗:")251        ):252            continue253        indent_here = len(line) - len(stripped)254        if current_file and (last_test_indent is None or indent_here < last_test_indent):255            describe_stack = [(i, d) for i, d in describe_stack if i < indent_here]256            describe_stack.append((indent_here, stripped))257    return out258 259 260def _detect_runner(test_cmds: str) -> str:261    joined = test_cmds.lower()262    if "pytest" in joined:263        return "pytest"264    if re.search(r"\bgo\s+test\b", joined):265        return "go"266    if re.search(r"\bcargo\s+test\b", joined):267        return "cargo"268    if any(k in joined for k in ("jest", "mocha", "vitest", "npm test", "yarn test", "pnpm test")):269        return "jest"270    return "unknown"271 272 273def parse_logs(runner: str, log: str) -> dict[str, str]:274    """Dispatch to the right per-runner parser. Empty dict if unknown."""275    if runner == "pytest":276        return parse_pytest(log)277    if runner == "go":278        return parse_go_test(log)279    if runner == "cargo":280        return parse_cargo_test(log)281    if runner == "jest":282        return parse_jest(log)283    return {}284 285 286# ---------------------------------------------------------------------------287# Grading288# ---------------------------------------------------------------------------289def _has_go_parent(name: str, status_map: dict[str, str]) -> bool:290    """Return True if a Go test has an ancestor represented in the status map."""291    parts = name.split("/")292 293    for i in range(1, len(parts)):294        parent = "/".join(parts[:i])295        if status_map.get(parent) == FAILED:296            return True297 298    return False299 300 301def grade(302    fail_to_pass: list[str],303    pass_to_pass: list[str],304    status_map: dict[str, str],305    runner: str | None = None,306) -> dict:307    """Compute the graded reward + strict resolved bool from a status map.308 309    f2p_rate   = (# F2P now PASSED) / (# F2P)310    p2p_rate   = (# P2P still PASSED) / (# P2P)   [1.0 when no P2P]311    reward     = f2p_rate * p2p_rate                       (dense training signal)312    resolved   = all F2P pass AND all P2P pass   (SWE-bench TRACKED resolution —313                 the gold patch always satisfies this, preserving the oracle314                 invariant)315 316    Two distinct EVAL signals (see the audit's "tracked vs command resolved"):317      * `resolved`         — tracked resolution (above). Gold patch -> True.318      * `command_resolved` — stricter: tracked resolution AND the selected test319                             command had NO failures outside F2P/P2P AND exit320                             code 0. Computed in main() where the exit code is321                             known. A benchmark that wants "the whole command322                             passed cleanly" gates on this; SWE-bench-style323                             scoring uses `resolved`.324 325    `untracked_failed` are FAILED tests in the run that are neither F2P nor326    P2P (e.g. always-failing/flaky tests pulled in by running a whole test327    file). They don't change the graded `reward` or tracked `resolved`, but328    they block `command_resolved` and are recorded for transparency.329    """330    f2p_total = len(fail_to_pass)331    p2p_total = len(pass_to_pass)332    f2p_set = set(fail_to_pass)333    p2p_set = set(pass_to_pass)334    f2p_passed = sum(1 for t in fail_to_pass if status_map.get(t) == PASSED)335    p2p_passed = sum(1 for t in pass_to_pass if status_map.get(t) == PASSED)336    # Tests that should have stayed green but regressed (PASS->not-pass).337    regressions = [t for t in pass_to_pass if status_map.get(t) != PASSED]338 339    # FAILED tests outside the tracked sets — the selected command isn't clean.340    if runner == "go":341        failed_tests = {342            t343            for t, s in status_map.items()344            if s == FAILED and t not in f2p_set and t not in p2p_set345        }346 347        untracked_failed = sorted(t for t in failed_tests if not _has_go_parent(t, status_map))348    else:349        untracked_failed = sorted(350            t351            for t, s in status_map.items()352            if s == FAILED and t not in f2p_set and t not in p2p_set353        )354 355    f2p_rate = (f2p_passed / f2p_total) if f2p_total else 0.0356    p2p_rate = (p2p_passed / p2p_total) if p2p_total else 1.0357    reward = f2p_rate * p2p_rate358    resolved = f2p_total > 0 and f2p_passed == f2p_total and p2p_passed == p2p_total359 360    return {361        "reward": round(max(0.0, min(1.0, reward)), 6),362        "resolved": resolved,363        "f2p_total": f2p_total,364        "f2p_passed": f2p_passed,365        "f2p_rate": round(f2p_rate, 6),366        "p2p_total": p2p_total,367        "p2p_passed": p2p_passed,368        "p2p_rate": round(p2p_rate, 6),369        "regressions": sorted(regressions),370        "untracked_failed_count": len(untracked_failed),371        "untracked_failed": untracked_failed[:20],  # cap the list372    }373 374 375# ---------------------------------------------------------------------------376# Entry point (invoked by tests/test.sh inside the container)377# ---------------------------------------------------------------------------378 379 380def _read_json_list(path: str) -> list[str]:381    try:382        with open(path, encoding="utf-8") as f:383            data = json.load(f)384        return [str(x) for x in data] if isinstance(data, list) else []385    except (OSError, ValueError):386        return []387 388 389def _read_text(path: str) -> str:390    try:391        with open(path, encoding="utf-8") as f:392            return f.read()393    except (OSError, UnicodeDecodeError):394        return ""395 396 397def main(argv: list[str] | None = None) -> int:398    p = argparse.ArgumentParser(description="pr_runtime graded F2P/P2P verifier")399    p.add_argument("--log", required=True, help="captured test-run log file")400    p.add_argument("--f2p", required=True, help="JSON file: FAIL_TO_PASS test names")401    p.add_argument("--p2p", required=True, help="JSON file: PASS_TO_PASS test names")402    p.add_argument("--runner", default="", help="pytest|go|cargo|jest (else auto-detect)")403    p.add_argument("--test-cmds", default="", help="test command string (runner auto-detect)")404    p.add_argument("--exit-code", type=int, default=1, help="test suite exit code (fallback)")405    p.add_argument("--out-dir", default="/logs/verifier", help="where to write reward.{txt,json}")406    args = p.parse_args(argv)407 408    log = _read_text(args.log)409    f2p = _read_json_list(args.f2p)410    p2p = _read_json_list(args.p2p)411    runner = args.runner.strip() or _detect_runner(args.test_cmds)412 413    status_map = parse_logs(runner, log)414 415    if not status_map:416        # Unparseable runner output → fall back to the binary exit-code reward417        # (a coarse TRAINING signal) so we never silently zero a real fix on an418        # unrecognized format. But `resolved` is the strict EVAL signal: without419        # parsed per-test status we have NO evidence the declared FAIL_TO_PASS420        # tests passed, so when an F2P oracle exists we must NOT claim resolved.421        # (resolved stays exit-code-based only when there's no declared oracle,422        # e.g. --skip-validation.)423        reward = 1.0 if args.exit_code == 0 else 0.0424        has_oracle = len(f2p) > 0425        resolved = (args.exit_code == 0) and not has_oracle426        breakdown = {427            "reward": reward,428            "resolved": resolved,429            "command_resolved": bool(resolved and args.exit_code == 0),430            "parse_status": "fallback_exitcode",431            "eval_trustworthy": not has_oracle,432            "runner": runner,433            "f2p_total": len(f2p),434            "p2p_total": len(p2p),435            "exit_code": args.exit_code,436        }437    else:438        breakdown = grade(f2p, p2p, status_map, runner)439        breakdown["parse_status"] = "ok"440        breakdown["runner"] = runner441        breakdown["tests_parsed"] = len(status_map)442        breakdown["exit_code"] = args.exit_code443        # Strict eval signal: tracked resolution AND a clean command (no444        # untracked failures, exit code 0). Benchmarks wanting "the whole test445        # command passed" gate on this; SWE-bench-style scoring uses `resolved`.446        breakdown["command_resolved"] = bool(447            breakdown["resolved"]448            and breakdown["untracked_failed_count"] == 0449            and args.exit_code == 0450        )451        reward = breakdown["reward"]452 453    os.makedirs(args.out_dir, exist_ok=True)454    with open(os.path.join(args.out_dir, "reward.txt"), "w", encoding="utf-8") as f:455        f.write(f"{reward:.6f}\n")456    with open(os.path.join(args.out_dir, "reward-details.json"), "w", encoding="utf-8") as f:457        json.dump(breakdown, f, indent=2)458 459    print(json.dumps(breakdown, indent=2))460    return 0461 462 463if __name__ == "__main__":464    sys.exit(main())465 466 467__all__ = [468    "grade",469    "main",470    "parse_cargo_test",471    "parse_go_test",472    "parse_jest",473    "parse_logs",474    "parse_pytest",475]476