jester1177/cloud-native-debug-env
0
1"""Determinism and score-range tests for the grader and environment."""2 3from server.environment import CICDDebugEnvironment4from server.graders import run_grader5from server.models import Action, ActionType, FileEdit6from server.tasks.task_registry import TASK_REGISTRY7 8 9# -- determinism --10 11 12def test_reset_deterministic_with_seed():13 """Same seed → same task, scenario, files, error."""14 env1 = CICDDebugEnvironment()15 env2 = CICDDebugEnvironment()16 17 obs1 = env1.reset(seed=42)18 obs2 = env2.reset(seed=42)19 20 assert obs1.task_id == obs2.task_id21 assert obs1.error.error_message == obs2.error.error_message22 assert [f.path for f in obs1.files] == [f.path for f in obs2.files]23 assert [f.content for f in obs1.files] == [f.content for f in obs2.files]24 25 26def test_grader_deterministic_same_trajectory():27 """Identical trajectory → identical score and breakdown."""28 trajectory = [29 {30 "step": 1,31 "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},32 "reward": 0.3,33 "done": False,34 "info": {"issues_fixed": 1, "issues_total": 2},35 },36 {37 "step": 2,38 "action": {"action_type": "submit"},39 "reward": 0.4,40 "done": True,41 "info": {"issues_fixed": 1, "issues_total": 2},42 },43 ]44 results = [run_grader("dockerfile_syntax", trajectory) for _ in range(10)]45 scores = [r.score for r in results]46 assert len(set(scores)) == 1, f"Non-deterministic scores: {scores}"47 breakdowns = [tuple(sorted(r.breakdown.items())) for r in results]48 assert len(set(breakdowns)) == 149 50 51def test_grader_deterministic_across_tasks():52 """Same trajectory structure scores identically regardless of task_id."""53 trajectory = [54 {55 "step": 1,56 "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},57 "reward": 0.3,58 "done": True,59 "info": {"issues_fixed": 1, "issues_total": 1},60 },61 ]62 scores = set()63 for task_id in TASK_REGISTRY:64 r = run_grader(task_id, trajectory)65 scores.add(r.score)66 # All tasks with same trajectory should get same score (task-agnostic grader)67 assert len(scores) == 1, f"Different scores across tasks: {scores}"68 69 70def test_full_episode_determinism():71 """Full episode replay produces identical trajectory and score."""72 scores = []73 for _ in range(5):74 env = CICDDebugEnvironment()75 env.reset(task_id="dockerfile_syntax", scenario_id="typo_filename")76 action = Action(77 action_type=ActionType.EDIT_FILE,78 edits=[FileEdit(file_path="Dockerfile", old_content="COPY requirments.txt .", new_content="COPY requirements.txt .")]79 )80 env.step(action)81 r = run_grader("dockerfile_syntax", env.trajectory)82 scores.append(r.score)83 assert len(set(scores)) == 1, f"Non-deterministic episode scores: {scores}"84 85 86# -- score ranges --87 88 89def test_empty_trajectory_scores_zero():90 r = run_grader("dockerfile_syntax", [])91 assert r.score == 0.092 assert r.steps_taken == 093 94 95def test_zero_fixes_scores_zero():96 trajectory = [97 {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},98 "reward": 0.0, "done": True, "info": {"issues_fixed": 0, "issues_total": 2}},99 ]100 r = run_grader("dockerfile_syntax", trajectory)101 assert r.score == 0.0102 103 104def test_partial_fix_scores_moderate():105 """1 of 2 issues fixed → score between 0.3 and 0.6."""106 trajectory = [107 {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},108 "reward": 0.3, "done": False, "info": {"issues_fixed": 1, "issues_total": 2}},109 {"step": 2, "action": {"action_type": "submit"},110 "reward": 0.0, "done": True, "info": {"issues_fixed": 1, "issues_total": 2}},111 ]112 r = run_grader("dockerfile_syntax", trajectory)113 assert 0.3 <= r.score <= 0.6, f"Partial fix score {r.score} out of range"114 115 116def test_complete_fix_scores_high():117 """All issues fixed → score >= 0.85."""118 trajectory = [119 {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},120 "reward": 0.3, "done": False, "info": {"issues_fixed": 1, "issues_total": 2}},121 {"step": 2, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},122 "reward": 0.3, "done": True, "info": {"issues_fixed": 2, "issues_total": 2}},123 ]124 r = run_grader("dockerfile_syntax", trajectory)125 assert r.score >= 0.85, f"Complete fix score {r.score} too low"126 127 128def test_perfect_score_achievable():129 """Single issue, single step → exactly 1.0."""130 trajectory = [131 {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},132 "reward": 0.3, "done": True, "info": {"issues_fixed": 1, "issues_total": 1}},133 ]134 r = run_grader("dockerfile_syntax", trajectory)135 assert r.score == 1.0, f"Perfect scenario scored {r.score}, not 1.0"136 137 138def test_hint_penalty_applied():139 """Hints reduce score by 0.05 each."""140 base_traj = [141 {"step": 1, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},142 "reward": 0.3, "done": True, "info": {"issues_fixed": 1, "issues_total": 1}},143 ]144 hint_traj = [145 {"step": 1, "action": {"action_type": "request_hint"}, "reward": -0.05, "done": False,146 "info": {"issues_fixed": 0, "issues_total": 1}},147 {"step": 2, "action": {"action_type": "edit_file", "edits": [{"file_path": "Dockerfile"}]},148 "reward": 0.3, "done": True, "info": {"issues_fixed": 1, "issues_total": 1}},149 ]150 r_base = run_grader("dockerfile_syntax", base_traj)151 r_hint = run_grader("dockerfile_syntax", hint_traj)152 assert r_base.score > r_hint.score153 assert abs((r_base.score - r_hint.score) - 0.08) < 0.05 # ~0.05 hint + efficiency decay154 155 156def test_score_always_in_0_1_range():157 """Score must always be between 0.0 and 1.0."""158 test_cases = [159 [],160 [{"step": 1, "action": {"action_type": "submit"}, "reward": 0.0, "done": True,161 "info": {"issues_fixed": 0, "issues_total": 5}}],162 # Many hints — could potentially go negative163 *[[{"step": i + 1, "action": {"action_type": "request_hint"}, "reward": -0.05, "done": i == 9,164 "info": {"issues_fixed": 0, "issues_total": 1}} for i in range(10)]],165 ]166 for traj in test_cases:167 r = run_grader("dockerfile_syntax", traj)168 assert 0.0 <= r.score <= 1.0, f"Score {r.score} out of [0, 1] range"169 170 171# -- difficulty progression --172 173 174def test_difficulty_progression():175 """Tasks are ordered by difficulty: easy < medium < hard."""176 difficulties = []177 for task_id, task_cls in TASK_REGISTRY.items():178 difficulties.append((task_id, task_cls.DIFFICULTY.value))179 180 expected_order = {181 "dockerfile_syntax": "easy",182 "dockerfile_runtime": "medium",183 "workflow_syntax_structure": "easy",184 "workflow_secrets_permissions": "medium",185 "ci_docker_integration": "medium",186 "multi_stage_pipeline_matrix": "hard",187 }188 for task_id, expected_diff in expected_order.items():189 actual = TASK_REGISTRY[task_id].DIFFICULTY.value190 assert actual == expected_diff, f"{task_id}: expected {expected_diff}, got {actual}"191 192 193def test_hard_tasks_have_more_issues():194 """Hard tasks should generally have more expected_fixes per scenario."""195 easy_max_issues = 0196 hard_min_issues = float("inf")197 198 for task_id, task_cls in TASK_REGISTRY.items():199 task = task_cls()200 for scenario in task.SCENARIOS:201 n_fixes = len(scenario["expected_fixes"])202 if task.DIFFICULTY.value == "easy":203 easy_max_issues = max(easy_max_issues, n_fixes)204 elif task.DIFFICULTY.value == "hard":205 hard_min_issues = min(hard_min_issues, n_fixes)206 207 # At least some hard scenarios should have more issues than easy ones208 assert hard_min_issues >= easy_max_issues, (209 f"Hard tasks ({hard_min_issues} min issues) should have >= issues than easy ({easy_max_issues} max)"210 )211 212 213def test_all_tasks_have_minimum_scenarios():214 """Each task must have at least 4 scenarios."""215 for task_id, task_cls in TASK_REGISTRY.items():216 assert len(task_cls.SCENARIOS) >= 4, f"{task_id} has only {len(task_cls.SCENARIOS)} scenarios (need >= 4)"217 218 219def test_scenario_ids_unique():220 """All scenario IDs must be unique within each task."""221 for task_id, task_cls in TASK_REGISTRY.items():222 ids = [s["id"] for s in task_cls.SCENARIOS]223 assert len(ids) == len(set(ids)), f"{task_id} has duplicate scenario IDs: {ids}"224 225 226def test_all_scenarios_have_required_fields():227 """Every scenario has id, files, error, expected_fixes."""228 for task_id, task_cls in TASK_REGISTRY.items():229 for scenario in task_cls.SCENARIOS:230 assert "id" in scenario, f"{task_id}: scenario missing 'id'"231 assert "files" in scenario, f"{task_id}/{scenario.get('id')}: missing 'files'"232 assert "error" in scenario, f"{task_id}/{scenario.get('id')}: missing 'error'"233 assert "expected_fixes" in scenario, f"{task_id}/{scenario.get('id')}: missing 'expected_fixes'"234 assert len(scenario["files"]) >= 1, f"{task_id}/{scenario['id']}: no files"235 assert len(scenario["expected_fixes"]) >= 1, f"{task_id}/{scenario['id']}: no expected_fixes"236 237 238# -- e2e grading --239 240 241def test_end_to_end_grading_all_tasks():242 """Every task/scenario can be reset, fixed, and graded with score > 0."""243 env = CICDDebugEnvironment()244 for task_id, task_cls in TASK_REGISTRY.items():245 task = task_cls()246 for scenario in task.SCENARIOS:247 obs = env.reset(task_id=task_id, scenario_id=scenario["id"])248 assert obs.total_issues >= 1249 assert obs.issues_fixed == 0250 # Just verify the grader doesn't crash on an empty trajectory251 r = run_grader(task_id, env.trajectory)252 assert r.score == 0.0253 