Team Ai
Apppublic

jester1177/cloudnative-devops-debug-env

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
__init__.py174 linesDownload Raw Back to graders
1"""Deterministic grader for trajectory scoring.2 3Scoring weights (difficulty-aware):4  base score      5%   (participation — guarantees score > 0)5  partial fixes  35%   (proportional to fix ratio)6  complete bonus 25%   (all issues fixed — scales with difficulty)7  efficiency     25%   (decays with extra steps — slower decay for harder tasks)8  hint penalty   -4%   each (reduced to -3% for hard/expert)9  failed edit    -2%   each10  difficulty     +5%   bonus for hard/expert tasks when fully solved11 12Score is clamped to [0.0, 1.0].13"""14 15from typing import Any, Dict, List16 17from server.models import GraderResult, TaskDifficulty18from server.tasks.task_registry import TASK_REGISTRY19 20# ── Base weights ──────────────────────────────────────────────21BASE_SCORE = 0.0522PARTIAL_FIX_WEIGHT = 0.3523COMPLETE_BONUS = 0.2524EFFICIENCY_MAX = 0.2525EFFICIENCY_DECAY = 0.03  # per extra step beyond optimal26HINT_PENALTY = 0.0427FAILED_ACTION_PENALTY = 0.0228 29# ── Difficulty modifiers ──────────────────────────────────────30# Maps difficulty → (complete_bonus_extra, efficiency_decay_mult, hint_penalty_mult)31#   complete_bonus_extra: added to COMPLETE_BONUS when all issues fixed32#   efficiency_decay_mult: multiplier on decay (lower = more forgiving)33#   hint_penalty_mult: multiplier on hint cost (lower = cheaper hints)34DIFFICULTY_MODIFIERS = {35    TaskDifficulty.EASY:   (0.00, 1.0, 1.0),36    TaskDifficulty.MEDIUM: (0.00, 0.9, 1.0),37    TaskDifficulty.HARD:   (0.03, 0.7, 0.75),38}39 40SCORE_FLOOR = 0.0141SCORE_CEIL = 0.9942 43EDIT_ACTION_TYPES = frozenset({44    "edit_file", "replace_line", "add_line",45    "delete_line", "add_block", "delete_block",46})47 48 49def _clamp(value: float) -> float:50    """Clamp score to [0, 1]."""51    return max(SCORE_FLOOR, min(SCORE_CEIL, round(value, 4)))52 53 54def _get_difficulty(task_id: str) -> TaskDifficulty:55    """Look up a task's difficulty from the registry."""56    task_cls = TASK_REGISTRY.get(task_id)57    if task_cls is None:58        return TaskDifficulty.MEDIUM59    return task_cls.DIFFICULTY60 61 62def run_grader(task_id: str, trajectory: List[Dict[str, Any]]) -> GraderResult:63    if task_id not in TASK_REGISTRY:64        raise ValueError(f"Unknown task: {task_id}")65 66    difficulty = _get_difficulty(task_id)67    bonus_extra, decay_mult, hint_mult = DIFFICULTY_MODIFIERS.get(68        difficulty, (0.00, 1.0, 1.0)69    )70 71    if not trajectory:72        return GraderResult(73            task_id=task_id,74            score=_clamp(BASE_SCORE),75            breakdown={76                "base": BASE_SCORE,77                "partial_fixes": 0.0,78                "complete_solution": 0.0,79                "efficiency": 0.0,80                "difficulty_bonus": 0.0,81                "hint_penalty": 0.0,82                "failed_action_penalty": 0.0,83            },84            feedback="No actions taken.",85            steps_taken=0,86            hints_used=0,87        )88 89    final_step = trajectory[-1]90    steps_taken = len(trajectory)91    hints_used = sum(92        1 for s in trajectory93        if s.get("action", {}).get("action_type") == "request_hint"94    )95 96    issues_fixed = int(final_step.get("info", {}).get("issues_fixed", 0))97    issues_total = max(1, int(final_step.get("info", {}).get("issues_total", 1)))98    fix_ratio = issues_fixed / issues_total99 100    # ── Component 1: Partial fix credit (proportional) ────────101    partial_score = PARTIAL_FIX_WEIGHT * fix_ratio102 103    # ── Component 2: Full-solution bonus ──────────────────────104    complete_bonus = COMPLETE_BONUS if issues_fixed == issues_total else 0.0105 106    # ── Component 3: Difficulty bonus ─────────────────────────107    # Extra reward for fully solving harder tasks108    diff_bonus = bonus_extra if issues_fixed == issues_total else 0.0109 110    # ── Component 4: Efficiency bonus ─────────────────────────111    # Harder tasks get slower decay (more forgiving on step count)112    if issues_fixed == 0:113        efficiency_score = 0.0114    elif steps_taken <= issues_total:115        efficiency_score = EFFICIENCY_MAX116    else:117        extra = steps_taken - issues_total118        effective_decay = EFFICIENCY_DECAY * decay_mult119        efficiency_score = max(0.0, EFFICIENCY_MAX - effective_decay * extra)120 121    # ── Component 5: Hint penalty ─────────────────────────────122    # Harder tasks get reduced hint penalty (hints are more reasonable)123    hint_pen = HINT_PENALTY * hint_mult * hints_used124 125    # ── Component 6: Failed action penalty ────────────────────126    failed_edits = 0127    for step in trajectory:128        action = step.get("action", {})129        if action.get("action_type") in EDIT_ACTION_TYPES:130            edits = action.get("edits") or []131            if not any(e.get("file_path") for e in edits):132                failed_edits += 1133    failed_pen = FAILED_ACTION_PENALTY * failed_edits134 135    raw = (136        BASE_SCORE137        + partial_score138        + complete_bonus139        + diff_bonus140        + efficiency_score141        - hint_pen142        - failed_pen143    )144    score = _clamp(raw)145 146    # ── Feedback ──────────────────────────────────────────────147    if score >= 0.85:148        feedback = "Excellent — all issues fixed efficiently."149    elif score >= 0.65:150        feedback = "Good job — most issues fixed."151    elif score >= 0.45:152        feedback = "Partial success — some issues remain."153    elif score >= 0.25:154        feedback = "Limited progress — review the error messages carefully."155    else:156        feedback = "Needs improvement — try analyzing the error phase first."157 158    return GraderResult(159        task_id=task_id,160        score=score,161        breakdown={162            "base": BASE_SCORE,163            "partial_fixes": round(partial_score, 4),164            "complete_solution": round(complete_bonus, 4),165            "difficulty_bonus": round(diff_bonus, 4),166            "efficiency": round(efficiency_score, 4),167            "hint_penalty": round(-hint_pen, 4),168            "failed_action_penalty": round(-failed_pen, 4),169        },170        feedback=feedback,171        steps_taken=steps_taken,172        hints_used=hints_used,173    )174