jester1177/cloudnative-devops-debug-env
0
1"""Deterministic grader for trajectory scoring.2 3Scoring weights (difficulty-aware):4 base score 5% (participation — guarantees score > 0)5 partial fixes 35% (proportional to fix ratio)6 complete bonus 25% (all issues fixed — scales with difficulty)7 efficiency 25% (decays with extra steps — slower decay for harder tasks)8 hint penalty -4% each (reduced to -3% for hard/expert)9 failed edit -2% each10 difficulty +5% bonus for hard/expert tasks when fully solved11 12Score is clamped to [0.0, 1.0].13"""14 15from typing import Any, Dict, List16 17from server.models import GraderResult, TaskDifficulty18from server.tasks.task_registry import TASK_REGISTRY19 20# ── Base weights ──────────────────────────────────────────────21BASE_SCORE = 0.0522PARTIAL_FIX_WEIGHT = 0.3523COMPLETE_BONUS = 0.2524EFFICIENCY_MAX = 0.2525EFFICIENCY_DECAY = 0.03 # per extra step beyond optimal26HINT_PENALTY = 0.0427FAILED_ACTION_PENALTY = 0.0228 29# ── Difficulty modifiers ──────────────────────────────────────30# Maps difficulty → (complete_bonus_extra, efficiency_decay_mult, hint_penalty_mult)31# complete_bonus_extra: added to COMPLETE_BONUS when all issues fixed32# efficiency_decay_mult: multiplier on decay (lower = more forgiving)33# hint_penalty_mult: multiplier on hint cost (lower = cheaper hints)34DIFFICULTY_MODIFIERS = {35 TaskDifficulty.EASY: (0.00, 1.0, 1.0),36 TaskDifficulty.MEDIUM: (0.00, 0.9, 1.0),37 TaskDifficulty.HARD: (0.03, 0.7, 0.75),38}39 40SCORE_FLOOR = 0.0141SCORE_CEIL = 0.9942 43EDIT_ACTION_TYPES = frozenset({44 "edit_file", "replace_line", "add_line",45 "delete_line", "add_block", "delete_block",46})47 48 49def _clamp(value: float) -> float:50 """Clamp score to [0, 1]."""51 return max(SCORE_FLOOR, min(SCORE_CEIL, round(value, 4)))52 53 54def _get_difficulty(task_id: str) -> TaskDifficulty:55 """Look up a task's difficulty from the registry."""56 task_cls = TASK_REGISTRY.get(task_id)57 if task_cls is None:58 return TaskDifficulty.MEDIUM59 return task_cls.DIFFICULTY60 61 62def run_grader(task_id: str, trajectory: List[Dict[str, Any]]) -> GraderResult:63 if task_id not in TASK_REGISTRY:64 raise ValueError(f"Unknown task: {task_id}")65 66 difficulty = _get_difficulty(task_id)67 bonus_extra, decay_mult, hint_mult = DIFFICULTY_MODIFIERS.get(68 difficulty, (0.00, 1.0, 1.0)69 )70 71 if not trajectory:72 return GraderResult(73 task_id=task_id,74 score=_clamp(BASE_SCORE),75 breakdown={76 "base": BASE_SCORE,77 "partial_fixes": 0.0,78 "complete_solution": 0.0,79 "efficiency": 0.0,80 "difficulty_bonus": 0.0,81 "hint_penalty": 0.0,82 "failed_action_penalty": 0.0,83 },84 feedback="No actions taken.",85 steps_taken=0,86 hints_used=0,87 )88 89 final_step = trajectory[-1]90 steps_taken = len(trajectory)91 hints_used = sum(92 1 for s in trajectory93 if s.get("action", {}).get("action_type") == "request_hint"94 )95 96 issues_fixed = int(final_step.get("info", {}).get("issues_fixed", 0))97 issues_total = max(1, int(final_step.get("info", {}).get("issues_total", 1)))98 fix_ratio = issues_fixed / issues_total99 100 # ── Component 1: Partial fix credit (proportional) ────────101 partial_score = PARTIAL_FIX_WEIGHT * fix_ratio102 103 # ── Component 2: Full-solution bonus ──────────────────────104 complete_bonus = COMPLETE_BONUS if issues_fixed == issues_total else 0.0105 106 # ── Component 3: Difficulty bonus ─────────────────────────107 # Extra reward for fully solving harder tasks108 diff_bonus = bonus_extra if issues_fixed == issues_total else 0.0109 110 # ── Component 4: Efficiency bonus ─────────────────────────111 # Harder tasks get slower decay (more forgiving on step count)112 if issues_fixed == 0:113 efficiency_score = 0.0114 elif steps_taken <= issues_total:115 efficiency_score = EFFICIENCY_MAX116 else:117 extra = steps_taken - issues_total118 effective_decay = EFFICIENCY_DECAY * decay_mult119 efficiency_score = max(0.0, EFFICIENCY_MAX - effective_decay * extra)120 121 # ── Component 5: Hint penalty ─────────────────────────────122 # Harder tasks get reduced hint penalty (hints are more reasonable)123 hint_pen = HINT_PENALTY * hint_mult * hints_used124 125 # ── Component 6: Failed action penalty ────────────────────126 failed_edits = 0127 for step in trajectory:128 action = step.get("action", {})129 if action.get("action_type") in EDIT_ACTION_TYPES:130 edits = action.get("edits") or []131 if not any(e.get("file_path") for e in edits):132 failed_edits += 1133 failed_pen = FAILED_ACTION_PENALTY * failed_edits134 135 raw = (136 BASE_SCORE137 + partial_score138 + complete_bonus139 + diff_bonus140 + efficiency_score141 - hint_pen142 - failed_pen143 )144 score = _clamp(raw)145 146 # ── Feedback ──────────────────────────────────────────────147 if score >= 0.85:148 feedback = "Excellent — all issues fixed efficiently."149 elif score >= 0.65:150 feedback = "Good job — most issues fixed."151 elif score >= 0.45:152 feedback = "Partial success — some issues remain."153 elif score >= 0.25:154 feedback = "Limited progress — review the error messages carefully."155 else:156 feedback = "Needs improvement — try analyzing the error phase first."157 158 return GraderResult(159 task_id=task_id,160 score=score,161 breakdown={162 "base": BASE_SCORE,163 "partial_fixes": round(partial_score, 4),164 "complete_solution": round(complete_bonus, 4),165 "difficulty_bonus": round(diff_bonus, 4),166 "efficiency": round(efficiency_score, 4),167 "hint_penalty": round(-hint_pen, 4),168 "failed_action_penalty": round(-failed_pen, 4),169 },170 feedback=feedback,171 steps_taken=steps_taken,172 hints_used=hints_used,173 )174 