Team Ai
Apppublic

jester1177/cloudnative-devops-debug-env

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
test_determinism.py253 linesDownload Raw Back to tests
1"""Determinism and score-range tests for the grader and environment."""2 3from server.environment import CloudNativeDebugEnvironment4from server.graders import run_grader5from server.models import Action, ActionType, FileEdit6from server.tasks.task_registry import TASK_REGISTRY7 8 9# -- determinism --10 11 12def test_reset_deterministic_with_seed():13    """Same seed → same task, scenario, files, error."""14    env1 = CloudNativeDebugEnvironment()15    env2 = CloudNativeDebugEnvironment()16 17    obs1 = env1.reset(seed=42)18    obs2 = env2.reset(seed=42)19 20    assert obs1.task_id == obs2.task_id21    assert obs1.error.error_message == obs2.error.error_message22    assert [f.path for f in obs1.files] == [f.path for f in obs2.files]23    assert [f.content for f in obs1.files] == [f.content for f in obs2.files]24 25 26def test_grader_deterministic_same_trajectory():27    """Identical trajectory → identical score and breakdown."""28    trajectory = [29        {30            "step": 1,31            "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},32            "reward": 0.3,33            "done": False,34            "info": {"issues_fixed": 1, "issues_total": 2},35        },36        {37            "step": 2,38            "action": {"action_type": "submit"},39            "reward": 0.4,40            "done": True,41            "info": {"issues_fixed": 1, "issues_total": 2},42        },43    ]44    results = [run_grader("dockerfile_syntax", trajectory) for _ in range(10)]45    scores = [r.score for r in results]46    assert len(set(scores)) == 1, f"Non-deterministic scores: {scores}"47    breakdowns = [tuple(sorted(r.breakdown.items())) for r in results]48    assert len(set(breakdowns)) == 149 50 51def test_grader_deterministic_across_tasks():52    """Same trajectory structure scores identically regardless of task_id."""53    trajectory = [54        {55            "step": 1,56            "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},57            "reward": 0.3,58            "done": True,59            "info": {"issues_fixed": 1, "issues_total": 1},60        },61    ]62    scores = set()63    for task_id in TASK_REGISTRY:64        r = run_grader(task_id, trajectory)65        scores.add(r.score)66    # All tasks with same trajectory should get same score (task-agnostic grader)67    assert len(scores) == 1, f"Different scores across tasks: {scores}"68 69 70def test_full_episode_determinism():71    """Full episode replay produces identical trajectory and score."""72    scores = []73    for _ in range(5):74        env = CloudNativeDebugEnvironment()75        env.reset(task_id="dockerfile_syntax", scenario_id="typo_filename")76        action = Action(77            action_type=ActionType.EDIT_FILE,78            edits=[FileEdit(file_path="Dockerfile", old_content="COPY requirments.txt .", new_content="COPY requirements.txt .")]79        )80        env.step(action)81        r = run_grader("dockerfile_syntax", env.trajectory)82        scores.append(r.score)83    assert len(set(scores)) == 1, f"Non-deterministic episode scores: {scores}"84 85 86# -- score ranges --87 88 89def test_empty_trajectory_scores_zero():90    r = run_grader("dockerfile_syntax", [])91    assert r.score == 0.092    assert r.steps_taken == 093 94 95def test_zero_fixes_scores_zero():96    trajectory = [97        {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},98         "reward": 0.0, "done": True, "info": {"issues_fixed": 0, "issues_total": 2}},99    ]100    r = run_grader("dockerfile_syntax", trajectory)101    assert r.score == 0.0102 103 104def test_partial_fix_scores_moderate():105    """1 of 2 issues fixed → score between 0.3 and 0.6."""106    trajectory = [107        {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},108         "reward": 0.3, "done": False, "info": {"issues_fixed": 1, "issues_total": 2}},109        {"step": 2, "action": {"action_type": "submit"},110         "reward": 0.0, "done": True, "info": {"issues_fixed": 1, "issues_total": 2}},111    ]112    r = run_grader("dockerfile_syntax", trajectory)113    assert 0.3 <= r.score <= 0.6, f"Partial fix score {r.score} out of range"114 115 116def test_complete_fix_scores_high():117    """All issues fixed → score >= 0.85."""118    trajectory = [119        {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},120         "reward": 0.3, "done": False, "info": {"issues_fixed": 1, "issues_total": 2}},121        {"step": 2, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},122         "reward": 0.3, "done": True, "info": {"issues_fixed": 2, "issues_total": 2}},123    ]124    r = run_grader("dockerfile_syntax", trajectory)125    assert r.score >= 0.85, f"Complete fix score {r.score} too low"126 127 128def test_perfect_score_achievable():129    """Single issue, single step → exactly 1.0."""130    trajectory = [131        {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},132         "reward": 0.3, "done": True, "info": {"issues_fixed": 1, "issues_total": 1}},133    ]134    r = run_grader("dockerfile_syntax", trajectory)135    assert r.score == 1.0, f"Perfect scenario scored {r.score}, not 1.0"136 137 138def test_hint_penalty_applied():139    """Hints reduce score by 0.05 each."""140    base_traj = [141        {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},142         "reward": 0.3, "done": True, "info": {"issues_fixed": 1, "issues_total": 1}},143    ]144    hint_traj = [145        {"step": 1, "action": {"action_type": "request_hint"}, "reward": -0.05, "done": False,146         "info": {"issues_fixed": 0, "issues_total": 1}},147        {"step": 2, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},148         "reward": 0.3, "done": True, "info": {"issues_fixed": 1, "issues_total": 1}},149    ]150    r_base = run_grader("dockerfile_syntax", base_traj)151    r_hint = run_grader("dockerfile_syntax", hint_traj)152    assert r_base.score > r_hint.score153    assert abs((r_base.score - r_hint.score) - 0.08) < 0.05  # ~0.05 hint + efficiency decay154 155 156def test_score_always_in_0_1_range():157    """Score must always be between 0.0 and 1.0."""158    test_cases = [159        [],160        [{"step": 1, "action": {"action_type": "submit"}, "reward": 0.0, "done": True,161          "info": {"issues_fixed": 0, "issues_total": 5}}],162        # Many hints — could potentially go negative163        *[[{"step": i + 1, "action": {"action_type": "request_hint"}, "reward": -0.05, "done": i == 9,164            "info": {"issues_fixed": 0, "issues_total": 1}} for i in range(10)]],165    ]166    for traj in test_cases:167        r = run_grader("dockerfile_syntax", traj)168        assert 0.0 <= r.score <= 1.0, f"Score {r.score} out of [0, 1] range"169 170 171# -- difficulty progression --172 173 174def test_difficulty_progression():175    """Tasks are ordered by difficulty: easy < medium < hard."""176    difficulties = []177    for task_id, task_cls in TASK_REGISTRY.items():178        difficulties.append((task_id, task_cls.DIFFICULTY.value))179 180    expected_order = {181        "dockerfile_syntax": "easy",182        "dockerfile_runtime": "medium",183        "workflow_syntax_structure": "easy",184        "workflow_secrets_permissions": "medium",185        "ci_docker_integration": "medium",186        "multi_stage_pipeline_matrix": "hard",187    }188    for task_id, expected_diff in expected_order.items():189        actual = TASK_REGISTRY[task_id].DIFFICULTY.value190        assert actual == expected_diff, f"{task_id}: expected {expected_diff}, got {actual}"191 192 193def test_hard_tasks_have_more_issues():194    """Hard tasks should generally have more expected_fixes per scenario."""195    easy_max_issues = 0196    hard_min_issues = float("inf")197 198    for task_id, task_cls in TASK_REGISTRY.items():199        task = task_cls()200        for scenario in task.SCENARIOS:201            n_fixes = len(scenario["expected_fixes"])202            if task.DIFFICULTY.value == "easy":203                easy_max_issues = max(easy_max_issues, n_fixes)204            elif task.DIFFICULTY.value == "hard":205                hard_min_issues = min(hard_min_issues, n_fixes)206 207    # At least some hard scenarios should have more issues than easy ones208    assert hard_min_issues >= easy_max_issues, (209        f"Hard tasks ({hard_min_issues} min issues) should have >= issues than easy ({easy_max_issues} max)"210    )211 212 213def test_all_tasks_have_minimum_scenarios():214    """Each task must have at least 4 scenarios."""215    for task_id, task_cls in TASK_REGISTRY.items():216        assert len(task_cls.SCENARIOS) >= 4, f"{task_id} has only {len(task_cls.SCENARIOS)} scenarios (need >= 4)"217 218 219def test_scenario_ids_unique():220    """All scenario IDs must be unique within each task."""221    for task_id, task_cls in TASK_REGISTRY.items():222        ids = [s["id"] for s in task_cls.SCENARIOS]223        assert len(ids) == len(set(ids)), f"{task_id} has duplicate scenario IDs: {ids}"224 225 226def test_all_scenarios_have_required_fields():227    """Every scenario has id, files, error, expected_fixes."""228    for task_id, task_cls in TASK_REGISTRY.items():229        for scenario in task_cls.SCENARIOS:230            assert "id" in scenario, f"{task_id}: scenario missing 'id'"231            assert "files" in scenario, f"{task_id}/{scenario.get('id')}: missing 'files'"232            assert "error" in scenario, f"{task_id}/{scenario.get('id')}: missing 'error'"233            assert "expected_fixes" in scenario, f"{task_id}/{scenario.get('id')}: missing 'expected_fixes'"234            assert len(scenario["files"]) >= 1, f"{task_id}/{scenario['id']}: no files"235            assert len(scenario["expected_fixes"]) >= 1, f"{task_id}/{scenario['id']}: no expected_fixes"236 237 238# -- e2e grading --239 240 241def test_end_to_end_grading_all_tasks():242    """Every task/scenario can be reset, fixed, and graded with score > 0."""243    env = CloudNativeDebugEnvironment()244    for task_id, task_cls in TASK_REGISTRY.items():245        task = task_cls()246        for scenario in task.SCENARIOS:247            obs = env.reset(task_id=task_id, scenario_id=scenario["id"])248            assert obs.total_issues >= 1249            assert obs.issues_fixed == 0250            # Just verify the grader doesn't crash on an empty trajectory251            r = run_grader(task_id, env.trajectory)252            assert r.score == 0.0253