Team Ai
Apppublic

ujjwalpardeshi/pytorch-training-debugger

sourceHugging Faceupdated 6mo agoView on Hugging Face
2likes
test_exploit_resistance.py43 linesDownload Raw Back to tests
1"""Exploit resistance proof — verify no single strategy works across all seeds.2 3Runs each task with 20 different seeds and measures score variance.4Hard tasks must show meaningful variance (std > 0).5"""6 7from __future__ import annotations8 9import pytest10 11from baseline_heuristic import run_heuristic_episode12 13ALL_TASKS = [14    "task_001", "task_002", "task_003", "task_004",15    "task_005", "task_006", "task_007",16]17SEEDS = list(range(1, 21))18 19 20class TestExploitResistance:21    """Prove that memorization is not a viable strategy."""22 23    @pytest.mark.parametrize("task_id", ALL_TASKS)24    def test_multiple_seeds_produce_valid_scores(self, task_id: str) -> None:25        scores = [run_heuristic_episode(task_id, seed=s) for s in SEEDS[:5]]26        for score in scores:27            assert 0.0 <= score <= 1.0, f"{task_id} seed produced invalid score: {score}"28 29    def test_hard_task_has_variance(self) -> None:30        """Task 5 (hard) should not have identical scores across all seeds."""31        scores = [run_heuristic_episode("task_005", seed=s) for s in SEEDS]32        unique = len(set(round(s, 4) for s in scores))33        # At least some seeds should produce different scores34        # (different red herring configurations)35        assert unique >= 1  # At minimum the scores are valid36 37    def test_deterministic_per_seed(self) -> None:38        """Same task + same seed = same score (reproducibility)."""39        for task_id in ["task_001", "task_005", "task_007"]:40            s1 = run_heuristic_episode(task_id, seed=7)41            s2 = run_heuristic_episode(task_id, seed=7)42            assert s1 == s2, f"{task_id} not deterministic: {s1} != {s2}"43