Team Ai
Apppublic

refaelz/legacy-code-challenge

sourceHugging Faceupdated 3d agoView on Hugging Face
0likes
student_interface.py2358 linesDownload Raw Back to root
1"""2Student Interface — Gradio GUI for the Legacy Code Challenge.3 4Two-page flow:5  Page 1 (Setup):     student name, GitHub URL, nesting level, num-bugs, options → Start6  Page 2 (Challenge): README, Code Editor, Test runner + AI Assistant sidebar7  Page 3 (Results):   Score breakdown, diffs, test results, hints review8"""9 10from __future__ import annotations11 12import difflib13import json14import os15import sys16import threading17import logging18from io import StringIO19from datetime import datetime20from pathlib import Path21 22import gradio as gr23 24# Configure detailed logging25logging.basicConfig(26    level=logging.INFO,27    format='%(asctime)s [%(levelname)s] %(name)s: %(message)s',28    handlers=[29        logging.FileHandler('app.log', encoding='utf-8'),30        logging.StreamHandler(sys.stdout)31    ]32)33logger = logging.getLogger(__name__)34 35from orchestrator.scoring import evaluate_submission36from orchestrator.hint_graph import get_hint37 38# Optional Google Sheets logging39try:40    from google_sheets_logger import diagnose_google_sheets, log_to_google_sheets41    GOOGLE_SHEETS_AVAILABLE = True42except ImportError:43    GOOGLE_SHEETS_AVAILABLE = False44 45# Optional Git submissions backup46try:47    from git_submissions_logger import diagnose_git_submissions, push_submission_to_git48    GIT_SUBMISSIONS_AVAILABLE = True49except ImportError:50    GIT_SUBMISSIONS_AVAILABLE = False51 52# Optional Git submissions backup53try:54    from git_submissions_logger import diagnose_git_submissions, push_submission_to_git55    GIT_SUBMISSIONS_AVAILABLE = True56except ImportError:57    GIT_SUBMISSIONS_AVAILABLE = False58 59 60# ── API Key Validation ────────────────────────────────────────────────────────61 62from llm_config import validate_api_key as validate_openai_key63 64 65def _google_sheets_configured() -> bool:66    return bool(os.getenv("GOOGLE_SHEET_ID")) and (67        bool(os.getenv("GOOGLE_SERVICE_ACCOUNT_JSON"))68        or Path("google-service-account.json").exists()69    )70 71 72def _git_submissions_configured() -> bool:73    return bool(os.getenv("SUBMISSIONS_REPO_URL"))74 75 76def _cloud_upload_capabilities() -> dict[str, dict[str, str | bool]]:77    google_configured = _google_sheets_configured()78    git_configured = _git_submissions_configured()79    return {80        "google_sheets": {81            "available": GOOGLE_SHEETS_AVAILABLE,82            "configured": google_configured,83            "reason": "" if google_configured else "Missing GOOGLE_SHEET_ID or service account credentials",84        },85        "git_backup": {86            "available": GIT_SUBMISSIONS_AVAILABLE,87            "configured": git_configured,88            "reason": "" if git_configured else "Missing SUBMISSIONS_REPO_URL",89        },90    }91 92 93# ── Challenge state loader ────────────────────────────────────────────────────94 95class ChallengeState:96    """Loads and exposes data from challenge_state.json."""97 98    def __init__(self, workspace_path: str) -> None:99        self.workspace = Path(workspace_path)100        state_file = self.workspace / "challenge_state.json"101        if not state_file.exists():102            raise FileNotFoundError(f"challenge_state.json not found in {workspace_path}")103        with open(state_file, encoding="utf-8") as f:104            data = json.load(f)105 106        self.github_url:       str  = data.get("github_url", "")107        # Normalise to forward slashes so comparisons work cross-platform108        raw_target = data.get("target_file", "")109        try:110            self.target_file: str = Path(raw_target).resolve().relative_to(self.workspace.resolve()).as_posix()111        except (ValueError, OSError):112            self.target_file = Path(raw_target).as_posix()113        self.original_code:    str  = data.get("original_code", "")114        self.sabotaged_code:   str  = data.get("sabotaged_code", "")115        self.function_name:    str  = data.get("function_name", "")116        self.bug_func_name:           str  = data.get("bug_func_name", "")117        self.original_bug_func_source: str  = data.get("original_bug_func_source", "")118        self.bug_func_names: list = data.get("bug_func_names", []) or (119            [self.bug_func_name] if self.bug_func_name else []120        )121        self.bug_func_sources_list: list = data.get("bug_func_sources_list", [])122        self.original_bug_func_sources_list: list = data.get("original_bug_func_sources_list", [])123        self.public_tests: list = data.get("public_tests", [])124        self.secret_tests: list = data.get("secret_tests", [])125        self.bug_description: str = data.get("bug_description", "")126        self.challenge_created_at: str = data.get("challenge_created_at", "")127        self.challenge_summary: str = data.get("challenge_summary", "")128        self.nesting_level:    int  = data.get("nesting_level", 3)129        self.refactoring_enabled: bool = data.get("refactoring_enabled", False)130        self.debug_mode:       bool = data.get("debug_mode", False)131 132        # sabotaged_files: {rel_posix_path: content} for every file the133        # saboteur touched.  Primary source: .metadata/ directory134        # written by the deployer immediately after sabotage (always up-to-date).135        # Fallback: sabotaged_files dict in JSON, then single sabotaged_code field.136        snapshot_dir = self.workspace / ".metadata"137        snap_files: dict[str, str] = {}138        if snapshot_dir.exists():139            # Get the target file path from the data for proper mapping140            target_rel_from_json = data.get("target_file", "")141            if target_rel_from_json:142                try:143                    # Convert to workspace-relative posix path144                    target_abs = Path(target_rel_from_json).resolve()145                    target_rel = target_abs.relative_to(self.workspace.resolve()).as_posix()146                except (ValueError, OSError):147                    target_rel = Path(target_rel_from_json).as_posix()148                149                # Find the corresponding snapshot file150                # The snapshot filename has separators replaced with __151                expected_snapshot_name = target_rel.replace("/", "__")152                snapshot_path = snapshot_dir / expected_snapshot_name153                154                if snapshot_path.exists():155                    try:156                        snap_files[target_rel] = snapshot_path.read_text(encoding="utf-8")157                    except Exception:158                        pass159 160        if snap_files:161            self.sabotaged_files: dict[str, str] = snap_files162        else:163            # Fall back to JSON fields164            stored = data.get("sabotaged_files", {})165            if not stored and self.sabotaged_code:166                try:167                    rel = Path(self.target_file).resolve().relative_to(168                        self.workspace.resolve()169                    ).as_posix()170                except ValueError:171                    rel = Path(self.target_file).name172                stored = {rel: self.sabotaged_code}173            self.sabotaged_files = stored174 175    @property176    def target_path(self) -> Path:177        return self.workspace / self.target_file178 179    def read_target(self) -> str:180        if self.target_path.exists():181            return self.target_path.read_text(encoding="utf-8")182        return self.sabotaged_code183 184    def write_target(self, code: str) -> None:185        self.target_path.write_text(code, encoding="utf-8")186 187    def write_py_file(self, rel_path: str, code: str) -> None:188        file_path = self.workspace / rel_path189        file_path.parent.mkdir(parents=True, exist_ok=True)190        file_path.write_text(code, encoding="utf-8")191 192    def reset_target(self) -> None:193        self.write_target(self.sabotaged_code)194 195    def list_py_files(self) -> list[str]:196        files = sorted(self.workspace.rglob("*.py"))197        return [198            f.relative_to(self.workspace).as_posix() for f in files199            if ".metadata" not in f.parts200        ]201 202    def read_py_file(self, rel_path: str) -> str:203        full = self.workspace / rel_path204        if full.exists():205            return full.read_text(encoding="utf-8")206        return f"# File not found: {rel_path}"207 208    def readme(self) -> str:209        readme_path = self.workspace / "STUDENT_README.md"210        if readme_path.exists():211            return readme_path.read_text(encoding="utf-8")212        return "# Challenge\n\nREADME not found."213 214    def challenge_info(self) -> dict:215        return {216            "function_name":                  self.function_name,217            "bug_func_name":                  self.bug_func_name,218            "target_file":                    self.target_file,219            "nesting_level":                  self.nesting_level,220            "refactoring_enabled":            self.refactoring_enabled,221            "debug_mode":                     self.debug_mode,222            "bug_func_names":                 self.bug_func_names,223            "bug_func_sources_list":          self.bug_func_sources_list,224            "original_bug_func_sources_list": self.original_bug_func_sources_list,225            "original_code":                  self.original_code,226        }227 228 229# ── Submission log ────────────────────────────────────────────────────────────230 231class SubmissionLog:232    """Persists submissions to <workspace>/submissions/ as JSON files."""233 234    def __init__(self, workspace: Path) -> None:235        self.log_dir = workspace / "submissions"236        self.log_dir.mkdir(exist_ok=True)237 238    def save(self, entry: dict) -> Path:239        submission_id = entry.get("submission_id") or datetime.now().strftime("%Y%m%d_%H%M%S_%f")240        path = self.log_dir / f"{submission_id}.json"241        with open(path, "w", encoding="utf-8") as f:242            json.dump(entry, f, indent=2, ensure_ascii=False)243        return path244 245    def update(self, path: Path, updates: dict) -> None:246        if not path.exists():247            return248        with open(path, encoding="utf-8") as f:249            entry = json.load(f)250        _deep_merge_dict(entry, updates)251        with open(path, "w", encoding="utf-8") as f:252            json.dump(entry, f, indent=2, ensure_ascii=False)253 254 255# ── Pipeline runner ───────────────────────────────────────────────────────────256 257def _run_pipeline(github_url: str, nesting_level: int, num_bugs: int, 258                  refactoring_enabled: bool = False, debug_mode: bool = False) -> str:259    """Invoke the architect pipeline and return the workspace path."""260    from architect.graph import build_graph261 262    graph = build_graph()263    result = graph.invoke({264        "github_url":          github_url,265        "nesting_level":       nesting_level,266        "refactoring_enabled": refactoring_enabled,267        "debug_mode":          debug_mode,268        "num_bugs":            num_bugs,269        "clone_path":          "",270        "target_file":         "",271        "original_code":       "",272        "sabotaged_code":      "",273        "function_name":       "",274        "test_args":           "",275        "expected_output":     "",276        "actual_output":       "",277        "bug_description":     "",278        "detailed_explanation": "",279        "challenge_summary":   "",280        "test_cases":          [],281        "public_tests":        [],282        "secret_tests":        [],283        "candidate_files":     [],284        "bug_func_name":       "",285        "bug_func_source":     "",286        "call_chain":          {},287    })288 289    workspace_path = result["clone_path"]290 291    # Read the actual file on disk (may differ from state if extra transforms ran)292    target_file_rel = result.get("target_file", "")293    actual_sabotaged_code = result.get("sabotaged_code", "")294    workspace_path_obj = Path(workspace_path).resolve()295    if target_file_rel:296        target_path = Path(workspace_path) / target_file_rel297        if target_path.exists():298            actual_sabotaged_code = target_path.read_text(encoding="utf-8")299 300    # Build sabotaged_files: {rel_posix_path: content} for every file the301    # saboteur modified.  Currently always one file (target_file).302    sabotaged_files: dict[str, str] = {}303    if target_file_rel and actual_sabotaged_code:304        try:305            rel_posix = Path(target_file_rel).resolve().relative_to(workspace_path_obj).as_posix()306        except (ValueError, OSError):307            rel_posix = Path(target_file_rel).as_posix()308        sabotaged_files[rel_posix] = actual_sabotaged_code309 310    # Persist challenge_state.json311    challenge_state = {312        "github_url":          github_url,313        "workspace_path":      workspace_path,314        "challenge_created_at": datetime.now().isoformat(),315        "target_file":         target_file_rel,316        "original_code":       result.get("original_code",    ""),317        "sabotaged_code":      actual_sabotaged_code,318        "sabotaged_files":     sabotaged_files,319        "function_name":       result.get("function_name",    ""),320        "bug_func_name":       result.get("bug_func_name",    ""),321        "bug_func_source":          result.get("bug_func_source",           ""),322        "original_bug_func_source": result.get("original_bug_func_source",  ""),323        "bug_func_names":              result.get("bug_func_names",              []),324        "bug_func_sources_list":       result.get("bug_func_sources_list",       []),325        "original_bug_func_sources_list": result.get("original_bug_func_sources_list", []),326        "test_cases":          result.get("test_cases",       []),327        "public_tests":        result.get("public_tests",     []),328        "secret_tests":        result.get("secret_tests",     []),329        "nesting_level":       nesting_level,330        "refactoring_enabled": refactoring_enabled,331        "debug_mode":          debug_mode,332        "bug_description":     result.get("bug_description",  ""),333        "challenge_summary":   result.get("challenge_summary", ""),334    }335    state_path = Path(workspace_path) / "challenge_state.json"336    with open(state_path, "w", encoding="utf-8") as f:337        json.dump(challenge_state, f, indent=2, ensure_ascii=False)338 339    return workspace_path340 341 342# ── Helpers ───────────────────────────────────────────────────────────────────343 344PENALTY_TABLE = [0, 2, 6, 12, 20, 30]345 346# Sets dark mode as the default on first visit and defines window.startTimer.347def _make_js(timer_minutes: int = 0) -> str:348    auto_start = (349        f"""350    (function tryStartTimer() {{351        if (document.getElementById('timer-inner')) {{352            window.startTimer({timer_minutes});353        }} else {{354            setTimeout(tryStartTimer, 150);355        }}356    }})();"""357        if timer_minutes > 0 else ""358    )359    return """() => {360    if (!localStorage.getItem('gradio-theme')) {361        localStorage.setItem('gradio-theme', 'dark');362    }363    window.startTimer = function(minutes) {364        if (!minutes || minutes <= 0) return;365        if (window._timerInterval) clearInterval(window._timerInterval);366        const endTime = Date.now() + minutes * 60 * 1000;367        function tick() {368            const remaining = Math.max(0, endTime - Date.now());369            const m = Math.floor(remaining / 60000);370            const s = Math.floor((remaining % 60000) / 1000);371            const el = document.getElementById('timer-inner');372            if (!el) return;373            if (remaining === 0) {374                el.innerHTML = '<span style="color:#ef4444;font-weight:bold;font-size:1.1em;">⏱️ TIME\\'S UP!</span>';375                clearInterval(window._timerInterval);376            } else {377                const color = remaining < 300000 ? '#ef4444' : (remaining < 600000 ? '#f59e0b' : '#22c55e');378                el.innerHTML = '⏱️ <span style="font-weight:bold;font-size:1.1em;color:' + color + ';">'379                    + String(m).padStart(2,'0') + ':' + String(s).padStart(2,'0') + '</span>';380            }381        }382        tick();383        window._timerInterval = setInterval(tick, 1000);384    };385""" + auto_start + "\n}"386 387 388_JS = _make_js(0)389 390_CSS = """391/* Keep footer visible */392footer svg { display: inline !important; }393.gradio-container { max-width: 100% !important; }394 395/* Setup page — centred card */396.setup-card { max-width: 700px; margin: 40px auto !important; }397 398/* Code editor — scrollable */399#code-editor .cm-scroller { overflow-y: auto !important; max-height: 60vh !important; }400#code-editor .cm-editor   { max-height: 60vh !important; }401 402/* Chatbot */403#ai-chatbot .wrap { height: 58vh !important; overflow-y: auto !important; }404 405/* Tab content */406.left-tabs .tabitem { overflow-y: auto !important; max-height: 80vh !important; }407 408/* Results diff blocks */409.diff-block { font-family: monospace; font-size: 0.85em; padding: 12px;410              border-radius: 8px; overflow-y: auto; max-height: 45vh;411              white-space: pre-wrap; background: #1e1e1e; color: #d4d4d4; }412 413/* Challenge timer — pushed to the right corner of the header */414.header-row { display: flex !important; align-items: center !important;415              width: 100% !important; gap: 8px; }416.header-row > * { flex-shrink: 0 !important; }417.header-row > *:first-child { flex: 1 1 auto !important; min-width: 0; }418#challenge-timer { flex: 0 0 auto !important; font-family: monospace;419                   font-size: 1em; text-align: right; padding: 4px 8px;420                   min-width: 90px; white-space: nowrap; }421"""422 423 424def _hint_md(hints_used: int, penalty: int) -> str:425    return (426        f"🤖 **AI Assistant** &nbsp;|&nbsp; "427        f"Hints used: **{hints_used}** &nbsp;|&nbsp; Penalty: **{penalty} pts** &nbsp; "428        f"_(−2 / −6 / −12 / −20 / −30)_"429    )430 431 432_HINT_DECLINES = {433    "no", "nope", "nah", "not now", "nevermind", "never mind",434    "no thanks", "no thank you", "skip", "cancel", "forget it",435    "don't", "dont", "no hint", "stop",436}437_CONFIRMATION_KWS = ["would you like", "shall i", "want me to", "proceed", "penalty to your score"]438 439 440def _is_decline(msg: str) -> bool:441    m = msg.strip().lower()442    return (443        m in _HINT_DECLINES444        or m.startswith("no ")445        or m.startswith("don't")446        or m.startswith("dont")447        or m.startswith("nope ")448        or m.startswith("nah ")449    )450 451 452def _is_confirmation_question(text: str) -> bool:453    t = text.lower()454    return any(kw in t for kw in _CONFIRMATION_KWS)455 456 457def _colorise_test_output(raw: str) -> str:458    """Wrap test output lines in coloured HTML spans."""459    html_lines = []460    for line in raw.split("\n"):461        if ": PASS" in line or "ALL PASS" in line:462            html_lines.append(f'<span style="color:#22c55e;font-weight:bold;">{line}</span>')463        elif ": FAIL" in line or ": CRASH" in line or "FAILED" in line:464            html_lines.append(f'<span style="color:#ef4444;font-weight:bold;">{line}</span>')465        else:466            html_lines.append(f'<span style="color:#d4d4d4;">{line}</span>')467    return (468        '<pre style="font-family:monospace;padding:12px;background:#1e1e1e;'469        'color:#d4d4d4;border-radius:8px;overflow-y:auto;max-height:60dvh;'470        'white-space:pre-wrap;">'471        + "<br>".join(html_lines)472        + "</pre>"473    )474 475 476def _normalize(code: str) -> str:477    """Strip trailing whitespace per line and normalize line endings."""478    return "\n".join(line.rstrip() for line in code.replace("\r\n", "\n").replace("\r", "\n").splitlines())479 480 481def _plain_diff_text(a: str, b: str, from_label: str, to_label: str) -> str:482    a_lines = _normalize(a).splitlines(keepends=True)483    b_lines = _normalize(b).splitlines(keepends=True)484    return "".join(difflib.unified_diff(a_lines, b_lines, fromfile=from_label, tofile=to_label, lineterm=""))485 486 487def _deep_merge_dict(target: dict, updates: dict) -> None:488    for key, value in updates.items():489        if isinstance(value, dict) and isinstance(target.get(key), dict):490            _deep_merge_dict(target[key], value)491        else:492            target[key] = value493 494 495def _workspace_diff_html(cs: "ChallengeState") -> str:496    """497    Show a unified diff for the target file the student changed.498    Compare the locally stored received-state snapshot vs the current disk content.499    This ensures only the student's changes are shown — not the sabotage noise.500    """501    workspace        = cs.workspace.resolve()502    sabotaged_files  = cs.sabotaged_files   # {rel_posix: received_content}503    sections: list[str] = []504 505    # Show diff for files modified by the saboteur (the target file)506    for rel_posix, received_content in sabotaged_files.items():507        full_path = workspace / rel_posix508        if not full_path.exists():509            continue510        current = full_path.read_text(encoding="utf-8")511        if _normalize(current) == _normalize(received_content):512            continue513        diff = _diff_html(received_content, current, rel_posix, rel_posix)514        sections.append(515            f"<h4 style='color:#94a3b8;margin:10px 0 4px;font-family:monospace;'>"516            f"📄 {rel_posix}</h4>" + diff517        )518 519    if not sections:520        return "<div style='color:#22c55e;padding:12px;'>No changes detected in workspace.</div>"521 522    return "".join(sections)523 524 525def _strip_comments_and_docstrings(code: str) -> str:526    """Remove comment-only lines and docstrings, keeping pure code."""527    import ast as _ast528    docstring_lines: set[int] = set()529    try:530        tree = _ast.parse(code)531        for node in _ast.walk(tree):532            if isinstance(node, (_ast.FunctionDef, _ast.AsyncFunctionDef,533                                  _ast.ClassDef, _ast.Module)):534                if (node.body535                        and isinstance(node.body[0], _ast.Expr)536                        and isinstance(node.body[0].value, _ast.Constant)537                        and isinstance(node.body[0].value.value, str)):538                    ds = node.body[0]539                    docstring_lines.update(range(ds.lineno, ds.end_lineno + 1))540    except Exception:541        pass542    result = []543    for i, line in enumerate(code.splitlines(), 1):544        if i in docstring_lines:545            continue546        if line.lstrip().startswith("#"):547            continue548        result.append(line)549    return "\n".join(result)550 551 552def _diff_html(a: str, b: str, from_label: str, to_label: str,553               strip_comments: bool = False) -> str:554    """Generate coloured unified-diff HTML between two code strings."""555    if strip_comments:556        a = _strip_comments_and_docstrings(a)557        b = _strip_comments_and_docstrings(b)558    a = _normalize(a)559    b = _normalize(b)560    a_lines = a.splitlines(keepends=True)561    b_lines = b.splitlines(keepends=True)562    diff = list(difflib.unified_diff(a_lines, b_lines,563                                     fromfile=from_label, tofile=to_label, lineterm=""))564    if not diff:565        return '<div class="diff-block" style="color:#22c55e;">No changes detected.</div>'566 567    html_lines = []568    for line in diff:569        escaped = line.replace("&", "&amp;").replace("<", "&lt;").replace(">", "&gt;")570        if line.startswith("+++") or line.startswith("---"):571            html_lines.append(f'<span style="color:#888;font-style:italic;">{escaped}</span><br>')572        elif line.startswith("@@"):573            html_lines.append(f'<span style="color:#60a5fa;">{escaped}</span><br>')574        elif line.startswith("+"):575            html_lines.append(f'<span style="color:#22c55e;background:#052e16;padding:2px 4px;">{escaped}</span><br>')576        elif line.startswith("-"):577            html_lines.append(f'<span style="color:#ef4444;background:#2d0a0a;padding:2px 4px;">{escaped}</span><br>')578        else:579            html_lines.append(f'<span style="color:#d4d4d4;">{escaped}</span><br>')580 581    return (582        '<div class="diff-block">'583        + "".join(html_lines)584        + "</div>"585    )586 587 588def _extract_function_source(code: str, func_name: str) -> str:589    """Extract a named function's source lines from code using ast."""590    import ast591    try:592        tree = ast.parse(code)593        for node in ast.walk(tree):594            if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):595                if node.name == func_name:596                    lines = code.splitlines()597                    return "\n".join(lines[node.lineno - 1 : node.end_lineno])598    except Exception:599        pass600    return ""601 602 603def _apply_one_bug_fix(604    current_code: str,605    func_name: str,606    orig_pre: str,607    sabot_pre: str,608) -> str | None:609    """610    Apply the fix for a single injected bug by surgically patching func_name611    inside current_code.  Returns the patched file content, or None if the612    fix could not be determined.613 614    Strategy: character-level diff between orig_pre (correct pre-transform615    source) and sabot_pre (buggy pre-transform source) surfaces the exact616    changed tokens.  Those tokens survive obfuscation so we can617    find-and-replace them directly in the fully-obfuscated current_code.618    """619    import difflib620 621    func_in_code = _extract_function_source(current_code, func_name)622    if not func_in_code:623        return None624 625    matcher = difflib.SequenceMatcher(None, orig_pre, sabot_pre, autojunk=False)626    opcodes = matcher.get_opcodes()627 628    fixed_func = func_in_code629    changed = False630 631    for tag, i1, i2, j1, j2 in opcodes:632        if tag == "equal":633            continue634 635        orig_frag  = orig_pre[i1:i2]636        sabot_frag = sabot_pre[j1:j2]637 638        if not orig_frag.strip() and not sabot_frag.strip():639            continue640 641        if tag in ("replace", "insert"):642            if not sabot_frag.strip():643                continue644 645            ctx_start_orig  = i1646            ctx_start_sabot = j1647            while ctx_start_orig > 0 and ctx_start_sabot > 0:648                co  = orig_pre[ctx_start_orig - 1]649                cs_ = sabot_pre[ctx_start_sabot - 1]650                if co != cs_ or not (co.isalnum() or co == "_"):651                    break652                ctx_start_orig  -= 1653                ctx_start_sabot -= 1654 655            search_str  = sabot_pre[ctx_start_sabot:j2]656            replace_str = orig_pre[ctx_start_orig:i2]657 658            if search_str and search_str in fixed_func:659                fixed_func = fixed_func.replace(search_str, replace_str, 1)660                changed = True661            elif sabot_frag and sabot_frag in fixed_func:662                fixed_func = fixed_func.replace(sabot_frag, orig_frag, 1)663                changed = True664 665        elif tag == "delete":666            if not orig_frag.strip():667                continue668 669            ctx_left      = sabot_pre[max(0, j1 - 30):j1]670            ctx_right_raw = sabot_pre[j2:j2 + 20]671            nl            = ctx_right_raw.find("\n")672            ctx_right     = ctx_right_raw[:nl] if nl != -1 else ctx_right_raw673 674            for trim in range(len(ctx_left)):675                search_str  = ctx_left[trim:] + ctx_right676                replace_str = ctx_left[trim:] + orig_frag + ctx_right677                if search_str and search_str in fixed_func:678                    fixed_func = fixed_func.replace(search_str, replace_str, 1)679                    changed = True680                    break681 682    if not changed or fixed_func == func_in_code:683        return None684 685    fixed = current_code.replace(func_in_code, fixed_func, 1)686    return fixed if fixed != current_code else None687 688 689def _compute_expected_fixed_code(cs: "ChallengeState") -> str | None:690    """691    Compute the expected correct version of the sabotaged file by applying692    surgical fixes for every injected bug.  Supports num_bugs > 1 by693    iterating over all (func_name, orig_pre, sabot_pre) triples stored in694    challenge_state.json.695    696    IMPORTANT: After inflate_hierarchy, the sources are POST-INFLATE (include wrappers).697    This ensures surgical patching works correctly even when wrapper layers exist.698    699    Returns None if no fix could be determined.700    """701    import json as _json702 703    try:704        _data = _json.loads((cs.workspace / "challenge_state.json").read_text(encoding="utf-8"))705    except Exception:706        _data = {}707 708    # Multi-bug lists (written by the new pipeline). Fall back to single-bug709    # fields for workspaces generated before this change.710    # NOTE: These are POST-INFLATE sources (updated by inflate_hierarchy)711    bug_func_names      = _data.get("bug_func_names") or cs.bug_func_names712    sabot_sources_list  = _data.get("bug_func_sources_list") or []713    orig_sources_list   = _data.get("original_bug_func_sources_list") or []714 715    # Single-bug fallback: wrap scalar fields in lists716    if not bug_func_names and cs.bug_func_name:717        bug_func_names = [cs.bug_func_name]718    if not sabot_sources_list:719        sabot_sources_list = [_data.get("bug_func_source", "")]720    if not orig_sources_list:721        orig_sources_list = [722            cs.original_bug_func_source723            or _extract_function_source(cs.original_code, cs.bug_func_name)724        ]725 726    if not bug_func_names:727        return None728 729    # Use the snapshot as the starting point (matches what the student received).730    try:731        target_rel = cs.target_path.resolve().relative_to(cs.workspace.resolve()).as_posix()732    except ValueError:733        target_rel = ""734    current_code = cs.sabotaged_files.get(target_rel) or cs.sabotaged_code735 736    any_changed = False737    for func_name, orig_pre, sabot_pre in zip(bug_func_names, orig_sources_list, sabot_sources_list):738        if not (orig_pre and sabot_pre):739            continue740        result = _apply_one_bug_fix(current_code, func_name, orig_pre, sabot_pre)741        if result is not None:742            current_code = result743            any_changed = True744 745    return current_code if any_changed else None746 747 748def _expected_fix_diff_html(cs: "ChallengeState", submitted_code: str = "") -> str:749    """750    Show two sections:751    1. What the expected fix looks like (buggy → expected).752    2. How the student's submission compares to the expected fix753       (expected → submitted): empty if perfect, otherwise shows extra/wrong changes.754    """755    func = cs.bug_func_name756 757    # Resolve the received (snapshot) content for the target file758    try:759        target_rel = cs.target_path.resolve().relative_to(cs.workspace.resolve()).as_posix()760    except ValueError:761        target_rel = ""762    received_code = cs.sabotaged_files.get(target_rel) or cs.sabotaged_code763 764    expected_fixed = _compute_expected_fixed_code(cs)765 766    # ── Fallback: function-level or full-file diff ────────────────────────────767    if expected_fixed is None:768        approx = received_code769        for fn in cs.bug_func_names:770            sabot_func = _extract_function_source(approx, fn)771            orig_func  = _extract_function_source(cs.original_code, fn)772            if sabot_func and orig_func:773                candidate = approx.replace(sabot_func, orig_func, 1)774                if candidate != approx:775                    approx = candidate776        if approx != received_code:777            expected_fixed = approx778        else:779            return _diff_html(received_code, cs.original_code, target_rel, target_rel,780                              strip_comments=True)781 782    # ── Section 1: expected fix ───────────────────────────────────────────────783    section1 = (784        "<h4 style='color:#94a3b8;margin:8px 0 4px;'>🎯 Expected fix</h4>"785        + _diff_html(received_code, expected_fixed, target_rel, target_rel,786                     strip_comments=True)787    )788 789    if not submitted_code:790        return section1791 792    # ── Section 2: workspace-wide comparison ─────────────────────────────────793    # "Perfect fix" = target file matches expected AND no other files were changed.794    def _ast_equivalent(a: str, b: str) -> bool:795        try:796            import ast as _ast797            return _ast.dump(_ast.parse(a)) == _ast.dump(_ast.parse(b))798        except SyntaxError:799            return _normalize(a) == _normalize(b)800 801    target_file_ok = _ast_equivalent(expected_fixed, submitted_code)802 803    # Check whether the student touched any files outside the sabotaged set.804    # Known-modified files = snapshot keys + the challenge target file itself.805    known_changed = set(cs.sabotaged_files.keys()) | ({target_rel} if target_rel else set())806    import subprocess as _sp807    try:808        proc = _sp.run(809            ["git", "diff", "HEAD", "--name-only"],810            cwd=str(cs.workspace.resolve()), capture_output=True,811            text=True, encoding="utf-8", timeout=10,812        )813        extra_changed = [814            f.replace("\\", "/") for f in proc.stdout.splitlines()815            if f.replace("\\", "/") not in known_changed816        ]817    except Exception:818        extra_changed = []819 820    is_perfect = target_file_ok and not extra_changed821 822    if is_perfect:823        verdict = (824            "<div style='margin-top:12px;padding:10px 14px;background:#052e16;"825            "border:1px solid #22c55e;border-radius:6px;color:#22c55e;font-weight:600;'>"826            "✅ Perfect fix — you changed exactly the right lines and nothing more."827            "</div>"828        )829    else:830        verdict = (831            "<div style='margin-top:12px;padding:10px 14px;background:#2d1a00;"832            "border:1px solid #f59e0b;border-radius:6px;color:#f59e0b;font-weight:600;'>"833            "⚠️ Your fix differs from the expected solution — see Your Changes on the left."834            "</div>"835        )836 837    return section1 + verdict838 839 840def _combined_changes_html(cs: "ChallengeState", submitted_code: str = "") -> str:841    """842    Render the "Changes & Expected" tab as a single HTML block.843 844    For each file in the union of (student-changed files ∪ expected-fix files),845    a row is rendered with two side-by-side diff panels:846      - Left:  student's diff (vs received sabotaged code)847      - Right: expected fix diff (vs received sabotaged code)848    If one side has no changes for a given file, a "No changes" placeholder is shown.849    A verdict banner spans the full width at the bottom.850    """851    import subprocess as _sp852 853    workspace = cs.workspace.resolve()854    try:855        target_rel = cs.target_path.resolve().relative_to(workspace).as_posix()856    except ValueError:857        target_rel = ""858 859    sabotaged_files = cs.sabotaged_files860    received_code   = sabotaged_files.get(target_rel) or cs.sabotaged_code861 862    # ── Collect student-changed files ─────────────────────────────────────────863    # student_changes: {rel_posix: (received_content, current_content)}864    student_changes: dict[str, tuple[str, str]] = {}865 866    for rel_posix, received_content in sabotaged_files.items():867        full_path = workspace / rel_posix868        if not full_path.exists():869            continue870        current = full_path.read_text(encoding="utf-8")871        if _normalize(current) != _normalize(received_content):872            student_changes[rel_posix] = (received_content, current)873 874    try:875        proc = _sp.run(876            ["git", "diff", "HEAD", "--name-only"],877            cwd=str(workspace), capture_output=True, text=True,878            encoding="utf-8", timeout=10,879        )880        for changed_rel in proc.stdout.splitlines():881            changed_posix = changed_rel.replace("\\", "/")882            if changed_posix in sabotaged_files:883                continue884            if not changed_rel.endswith(".py"):885                continue886            orig_proc = _sp.run(887                ["git", "show", f"HEAD:{changed_rel}"],888                cwd=str(workspace), capture_output=True, text=True,889                encoding="utf-8", timeout=10,890            )891            if orig_proc.returncode != 0:892                continue893            full_path = workspace / changed_rel894            if not full_path.exists():895                continue896            current = full_path.read_text(encoding="utf-8")897            if _normalize(current) != _normalize(orig_proc.stdout):898                student_changes[changed_posix] = (orig_proc.stdout, current)899    except Exception:900        pass901 902    # ── Compute expected fix for target file ──────────────────────────────────903    expected_fixed = _compute_expected_fixed_code(cs)904    if expected_fixed is None and cs.bug_func_name:905        sabot_func = _extract_function_source(received_code, cs.bug_func_name)906        orig_func  = _extract_function_source(cs.original_code, cs.bug_func_name)907        if sabot_func and orig_func:908            ef = received_code.replace(sabot_func, orig_func, 1)909            if ef != received_code:910                expected_fixed = ef911 912    # ── Build union of files to show ──────────────────────────────────────────913    expected_files = {target_rel} if (target_rel and expected_fixed is not None) else set()914    all_files = sorted(set(student_changes.keys()) | expected_files)915 916    # ── Column header row ─────────────────────────────────────────────────────917    col_headers = (918        "<div style='display:flex;gap:12px;margin-bottom:8px;'>"919        "<div style='flex:1;min-width:0;font-weight:600;color:#cbd5e1;'>✏️ Your Changes"920        " <span style='font-weight:400;color:#64748b;font-size:0.88em;'>"921        "— your submitted code vs the buggy code you received</span></div>"922        "<div style='flex:1;min-width:0;font-weight:600;color:#cbd5e1;'>🎯 Expected Fix"923        " <span style='font-weight:400;color:#64748b;font-size:0.88em;'>"924        "— the minimal correct fix</span></div>"925        "</div>"926    )927 928    if not all_files:929        return col_headers + "<div class='diff-block' style='color:#22c55e;'>No changes detected in workspace.</div>"930 931    _no_change_left  = "<div class='diff-block' style='color:#22c55e;'>No changes</div>"932    _no_change_right = "<div class='diff-block' style='color:#64748b;'>No change expected</div>"933 934    # ── Per-file side-by-side rows ────────────────────────────────────────────935    rows: list[str] = []936    for rel in all_files:937        file_header = (938            f"<h4 style='color:#94a3b8;margin:6px 0 4px;font-family:monospace;'>📄 {rel}</h4>"939        )940 941        # Left: student's changes942        if rel in student_changes:943            recv, curr = student_changes[rel]944            left_body = _diff_html(recv, curr, rel, rel)945        else:946            left_body = _no_change_left947 948        # Right: expected fix (only for the sabotaged target file)949        if rel == target_rel and expected_fixed is not None:950            right_body = _diff_html(received_code, expected_fixed, rel, rel)951        else:952            right_body = _no_change_right953 954        rows.append(955            "<div style='display:flex;gap:12px;margin-bottom:20px;'>"956            f"<div style='flex:1;min-width:0;'>{file_header}{left_body}</div>"957            f"<div style='flex:1;min-width:0;'>{file_header}{right_body}</div>"958            "</div>"959        )960 961    # ── Verdict banner (full-width) ───────────────────────────────────────────962    verdict = ""963    if submitted_code:964        def _ast_eq(a: str, b: str) -> bool:965            try:966                import ast as _ast967                return _ast.dump(_ast.parse(a)) == _ast.dump(_ast.parse(b))968            except SyntaxError:969                return _normalize(a) == _normalize(b)970 971        target_ok    = expected_fixed is not None and _ast_eq(expected_fixed, submitted_code)972        extra_files  = [f for f in student_changes if f != target_rel]973        is_perfect   = target_ok and not extra_files974 975        if is_perfect:976            verdict = (977                "<div style='margin-top:12px;padding:10px 14px;background:#052e16;"978                "border:1px solid #22c55e;border-radius:6px;color:#22c55e;font-weight:600;'>"979                "✅ Perfect fix — you changed exactly the right lines and nothing more."980                "</div>"981            )982        else:983            verdict = (984                "<div style='margin-top:12px;padding:10px 14px;background:#2d1a00;"985                "border:1px solid #f59e0b;border-radius:6px;color:#f59e0b;font-weight:600;'>"986                "⚠️ Your fix differs from the expected solution."987                "</div>"988            )989 990    return col_headers + "".join(rows) + verdict991 992 993def _hints_html(hint_log: list) -> str:994    """Render hint log with the student's actual question and the assistant response."""995 996    def _esc(s: str) -> str:997        return (s.replace("&", "&amp;").replace("<", "&lt;").replace(">", "&gt;"))998 999    entries = [e for e in (hint_log or []) if isinstance(e, dict)]1000 1001    if not entries:1002        return "<p style='color:#888;font-style:italic;'>No hints were used.</p>"1003 1004    parts: list[str] = []1005    for n, entry in enumerate(entries, 1):1006        question = entry.get("question", "").strip() or entry.get("summary", "").strip()1007        response = entry.get("response", "").strip()1008        hint_summary = entry.get("hint_summary", "").strip()1009        if not question:1010            first_line = response.split("\n")[0].strip()1011            dot = first_line.find(". ")1012            question = first_line[: dot + 1] if dot != -1 else first_line1013        parts.append(1014            f'<div style="margin:6px 0;padding:8px 14px;border:1px solid #374151;'1015            f'border-radius:8px;background:#1a2e1a;">'1016            f'<span style="color:#93c5fd;font-weight:bold;margin-right:10px;">Hint #{n}</span>'1017            f'<div style="color:#86efac;margin-top:6px;"><strong>Question:</strong> {_esc(question)}</div>'1018            f'<div style="color:#d1d5db;margin-top:6px;white-space:pre-wrap;"><strong>Answer:</strong> {_esc(response)}</div>'1019            + (f'<div style="color:#94a3b8;margin-top:6px;font-size:0.9em;"><strong>Internal hint summary:</strong> {_esc(hint_summary)}</div>' if hint_summary else '') +1020            f'</div>'1021        )1022 1023    header = (f'<p style="color:#888;font-size:0.85em;margin:0 0 8px 0;">'1024              f'{len(entries)} hint(s) used</p>')1025    return (1026        header1027        + '<div style="max-height:50vh;overflow-y:auto;padding:4px;">'1028        + "".join(parts)1029        + "</div>"1030    )1031 1032 1033def _score_summary_html(result: dict) -> str:1034    """Build a prominent score summary block."""1035    total       = result["total_score"]1036    llm_score   = result.get("llm_score", total)1037    explanation = result.get("llm_explanation", "")1038    penalty     = result["hint_penalty"]1039    passed      = result["passed"]1040    ttl         = result["total_tests"]1041    all_ok      = result["all_passed"]1042 1043    color = "#22c55e" if all_ok else ("#f59e0b" if total >= 50 else "#ef4444")1044    badge = "🎉 ALL TESTS PASS!" if all_ok else ("⚠️ Partial" if total > 0 else "❌ Failed")1045 1046    explanation_html = ""1047    if explanation:1048        import html as _html1049        escaped = _html.escape(explanation)1050        explanation_html = (1051            f'<div style="margin-top:14px;padding:10px 14px;background:#0f172a;'1052            f'border-left:3px solid #60a5fa;border-radius:4px;color:#cbd5e1;'1053            f'font-size:0.9em;line-height:1.5;">'1054            f'<strong style="color:#60a5fa;">🤖 AI Evaluation:</strong><br>{escaped}</div>'1055        )1056 1057    return f"""1058<div style="padding:20px;border-radius:12px;background:#1e1e1e;border:2px solid {color};">1059  <div style="font-size:2.5em;font-weight:bold;color:{color};text-align:center;">{total}/100</div>1060  <div style="text-align:center;color:{color};font-size:1.1em;margin-bottom:16px;">{badge}</div>1061  <table style="width:100%;border-collapse:collapse;font-family:monospace;">1062    <tr>1063      <td style="padding:6px 12px;color:#d4d4d4;">🤖 AI Score</td>1064      <td style="padding:6px 12px;color:#60a5fa;text-align:right;font-weight:bold;">{llm_score}/100</td>1065      <td style="padding:6px 12px;color:#888;">({passed}/{ttl} tests passed)</td>1066    </tr>1067    <tr>1068      <td style="padding:6px 12px;color:#d4d4d4;">💡 Hint Penalty</td>1069      <td style="padding:6px 12px;color:#f87171;text-align:right;font-weight:bold;">−{penalty}</td>1070      <td style="padding:6px 12px;color:#888;"></td>1071    </tr>1072  </table>1073  {explanation_html}1074</div>1075"""1076 1077 1078def _submission_outcome(result: dict) -> str:1079    if not result.get("touched_bug_function", True):1080        return "wrong_location" if result.get("changed_function_names") else "no_attempt"1081    if result.get("all_passed"):1082        return "full_fix"1083    before_passed = result.get("before_passed", 0)1084    after_passed = result.get("passed", 0)1085    if after_passed > before_passed:1086        return "partial_fix"1087    if after_passed < before_passed:1088        return "regression"1089    return "attempted_no_improvement"1090 1091 1092def _build_submission_record(1093    cs: "ChallengeState",1094    student_name: str,1095    submitted_code: str,1096    result: dict,1097    hints_used: int,1098    chat_log: list,1099    hint_log: list,1100    attempt_number: int,1101    debug_logs: str,1102    enable_cloud_uploads: bool,1103) -> dict:1104    timestamp = datetime.now()1105    submission_id = timestamp.strftime("%Y%m%d_%H%M%S_%f")1106    expected_fixed_code = _compute_expected_fixed_code(cs) or cs.original_code1107    challenge_id = f"{cs.workspace.name}:{cs.target_file}:{cs.function_name or 'unknown'}"1108    student_name = (student_name or "Anonymous").strip() or "Anonymous"1109    outcome = _submission_outcome(result)1110 1111    cloud_caps = _cloud_upload_capabilities()1112    google_status = "pending" if enable_cloud_uploads and cloud_caps["google_sheets"]["available"] and cloud_caps["google_sheets"]["configured"] else "disabled"1113    git_status = "pending" if enable_cloud_uploads and cloud_caps["git_backup"]["available"] and cloud_caps["git_backup"]["configured"] else "disabled"1114    google_error = "" if google_status == "pending" else str(cloud_caps["google_sheets"]["reason"])1115    git_error = "" if git_status == "pending" else str(cloud_caps["git_backup"]["reason"])1116 1117    return {1118        "submission_id": submission_id,1119        "timestamp": timestamp.isoformat(),1120        "student": {1121            "name": student_name,1122            "email": "",1123        },1124        "challenge": {1125            "challenge_id": challenge_id,1126            "created_at": cs.challenge_created_at,1127            "challenge_summary": cs.challenge_summary,1128            "workspace_name": cs.workspace.name,1129            "workspace_path": str(cs.workspace),1130            "repo_url": cs.github_url,1131            "target_file": cs.target_file,1132            "function_name": cs.function_name,1133            "bug_func_name": cs.bug_func_name,1134            "bug_func_names": cs.bug_func_names,1135            "bug_description": cs.bug_description,1136            "difficulty_level": cs.nesting_level,1137            "refactoring_enabled": cs.refactoring_enabled,1138            "debug_mode": cs.debug_mode,1139            "challenge_prompt": cs.readme(),1140            "original_code": cs.original_code,1141            "buggy_code": cs.sabotaged_code,1142            "expected_fixed_code": expected_fixed_code,1143            "public_tests": cs.public_tests,1144            "secret_tests": cs.secret_tests,1145        },1146        "submission": {1147            "attempt_number": attempt_number,1148            "hints_used": hints_used,1149            "chat_history": chat_log or [],1150            "hint_history": hint_log or [],1151            "student_code": submitted_code,1152            "student_diff": _plain_diff_text(1153                cs.sabotaged_code,1154                submitted_code,1155                f"received/{cs.target_file}",1156                f"student/{cs.target_file}",1157            ),1158            "changed_function_names": result.get("changed_function_names", []),1159            "changed_functions": result.get("changed_functions", []),1160            "bug_function_names": result.get("bug_function_names", []),1161            "touched_bug_function": result.get("touched_bug_function", False),1162        },1163        "tests": {1164            "before_passed": result.get("before_passed", 0),1165            "before_total": result.get("before_total", 0),1166            "public_passed": result.get("public_passed", 0),1167            "public_total": result.get("public_total", 0),1168            "secret_passed": result.get("secret_passed", 0),1169            "secret_total": result.get("secret_total", 0),1170            "passed_tests": result.get("passed", 0),1171            "total_tests": result.get("total_tests", 0),1172            "all_passed": result.get("all_passed", False),1173            "per_test_before": result.get("per_test_before", []),1174            "per_test_after": result.get("per_test_after", []),1175            "public_output": result.get("public_output", ""),1176            "secret_output": result.get("secret_output", ""),1177            "test_output": result.get("test_output", ""),1178        },1179        "scoring": {1180            "llm_score": result.get("llm_score", result.get("total_score", 0)),1181            "llm_explanation": result.get("llm_explanation", ""),1182            "hint_penalty": result.get("hint_penalty", 0),1183            "total_score": result.get("total_score", 0),1184            "test_score": result.get("test_score", 0),1185            "correct_location": result.get("correct_location", False),1186            "submission_outcome": outcome,1187        },1188        "system": {1189            "llm_provider": os.getenv("LLM_PROVIDER", ""),1190            "llm_model": os.getenv("LLM_MODEL", ""),1191            "agent_model": os.getenv("AGENT_MODEL", ""),1192        },1193        "cloud": {1194            "google_sheets": {"status": google_status, "error": google_error},1195            "git_backup": {"status": git_status, "error": git_error},1196        },1197        "debug": {1198            "captured_logs": debug_logs,1199        },1200    }

Showing the first 1,200 of 2358 lines. Download the file for the rest.