refaelz/legacy-code-challenge
0
1"""2Student Interface — Gradio GUI for the Legacy Code Challenge.3 4Two-page flow:5 Page 1 (Setup): student name, GitHub URL, nesting level, num-bugs, options → Start6 Page 2 (Challenge): README, Code Editor, Test runner + AI Assistant sidebar7 Page 3 (Results): Score breakdown, diffs, test results, hints review8"""9 10from __future__ import annotations11 12import difflib13import json14import os15import sys16import threading17import logging18from io import StringIO19from datetime import datetime20from pathlib import Path21 22import gradio as gr23 24# Configure detailed logging25logging.basicConfig(26 level=logging.INFO,27 format='%(asctime)s [%(levelname)s] %(name)s: %(message)s',28 handlers=[29 logging.FileHandler('app.log', encoding='utf-8'),30 logging.StreamHandler(sys.stdout)31 ]32)33logger = logging.getLogger(__name__)34 35from orchestrator.scoring import evaluate_submission36from orchestrator.hint_graph import get_hint37 38# Optional Google Sheets logging39try:40 from google_sheets_logger import diagnose_google_sheets, log_to_google_sheets41 GOOGLE_SHEETS_AVAILABLE = True42except ImportError:43 GOOGLE_SHEETS_AVAILABLE = False44 45# Optional Git submissions backup46try:47 from git_submissions_logger import diagnose_git_submissions, push_submission_to_git48 GIT_SUBMISSIONS_AVAILABLE = True49except ImportError:50 GIT_SUBMISSIONS_AVAILABLE = False51 52# Optional Git submissions backup53try:54 from git_submissions_logger import diagnose_git_submissions, push_submission_to_git55 GIT_SUBMISSIONS_AVAILABLE = True56except ImportError:57 GIT_SUBMISSIONS_AVAILABLE = False58 59 60# ── API Key Validation ────────────────────────────────────────────────────────61 62from llm_config import validate_api_key as validate_openai_key63 64 65def _google_sheets_configured() -> bool:66 return bool(os.getenv("GOOGLE_SHEET_ID")) and (67 bool(os.getenv("GOOGLE_SERVICE_ACCOUNT_JSON"))68 or Path("google-service-account.json").exists()69 )70 71 72def _git_submissions_configured() -> bool:73 return bool(os.getenv("SUBMISSIONS_REPO_URL"))74 75 76def _cloud_upload_capabilities() -> dict[str, dict[str, str | bool]]:77 google_configured = _google_sheets_configured()78 git_configured = _git_submissions_configured()79 return {80 "google_sheets": {81 "available": GOOGLE_SHEETS_AVAILABLE,82 "configured": google_configured,83 "reason": "" if google_configured else "Missing GOOGLE_SHEET_ID or service account credentials",84 },85 "git_backup": {86 "available": GIT_SUBMISSIONS_AVAILABLE,87 "configured": git_configured,88 "reason": "" if git_configured else "Missing SUBMISSIONS_REPO_URL",89 },90 }91 92 93# ── Challenge state loader ────────────────────────────────────────────────────94 95class ChallengeState:96 """Loads and exposes data from challenge_state.json."""97 98 def __init__(self, workspace_path: str) -> None:99 self.workspace = Path(workspace_path)100 state_file = self.workspace / "challenge_state.json"101 if not state_file.exists():102 raise FileNotFoundError(f"challenge_state.json not found in {workspace_path}")103 with open(state_file, encoding="utf-8") as f:104 data = json.load(f)105 106 self.github_url: str = data.get("github_url", "")107 # Normalise to forward slashes so comparisons work cross-platform108 raw_target = data.get("target_file", "")109 try:110 self.target_file: str = Path(raw_target).resolve().relative_to(self.workspace.resolve()).as_posix()111 except (ValueError, OSError):112 self.target_file = Path(raw_target).as_posix()113 self.original_code: str = data.get("original_code", "")114 self.sabotaged_code: str = data.get("sabotaged_code", "")115 self.function_name: str = data.get("function_name", "")116 self.bug_func_name: str = data.get("bug_func_name", "")117 self.original_bug_func_source: str = data.get("original_bug_func_source", "")118 self.bug_func_names: list = data.get("bug_func_names", []) or (119 [self.bug_func_name] if self.bug_func_name else []120 )121 self.bug_func_sources_list: list = data.get("bug_func_sources_list", [])122 self.original_bug_func_sources_list: list = data.get("original_bug_func_sources_list", [])123 self.public_tests: list = data.get("public_tests", [])124 self.secret_tests: list = data.get("secret_tests", [])125 self.bug_description: str = data.get("bug_description", "")126 self.challenge_created_at: str = data.get("challenge_created_at", "")127 self.challenge_summary: str = data.get("challenge_summary", "")128 self.nesting_level: int = data.get("nesting_level", 3)129 self.refactoring_enabled: bool = data.get("refactoring_enabled", False)130 self.debug_mode: bool = data.get("debug_mode", False)131 132 # sabotaged_files: {rel_posix_path: content} for every file the133 # saboteur touched. Primary source: .metadata/ directory134 # written by the deployer immediately after sabotage (always up-to-date).135 # Fallback: sabotaged_files dict in JSON, then single sabotaged_code field.136 snapshot_dir = self.workspace / ".metadata"137 snap_files: dict[str, str] = {}138 if snapshot_dir.exists():139 # Get the target file path from the data for proper mapping140 target_rel_from_json = data.get("target_file", "")141 if target_rel_from_json:142 try:143 # Convert to workspace-relative posix path144 target_abs = Path(target_rel_from_json).resolve()145 target_rel = target_abs.relative_to(self.workspace.resolve()).as_posix()146 except (ValueError, OSError):147 target_rel = Path(target_rel_from_json).as_posix()148 149 # Find the corresponding snapshot file150 # The snapshot filename has separators replaced with __151 expected_snapshot_name = target_rel.replace("/", "__")152 snapshot_path = snapshot_dir / expected_snapshot_name153 154 if snapshot_path.exists():155 try:156 snap_files[target_rel] = snapshot_path.read_text(encoding="utf-8")157 except Exception:158 pass159 160 if snap_files:161 self.sabotaged_files: dict[str, str] = snap_files162 else:163 # Fall back to JSON fields164 stored = data.get("sabotaged_files", {})165 if not stored and self.sabotaged_code:166 try:167 rel = Path(self.target_file).resolve().relative_to(168 self.workspace.resolve()169 ).as_posix()170 except ValueError:171 rel = Path(self.target_file).name172 stored = {rel: self.sabotaged_code}173 self.sabotaged_files = stored174 175 @property176 def target_path(self) -> Path:177 return self.workspace / self.target_file178 179 def read_target(self) -> str:180 if self.target_path.exists():181 return self.target_path.read_text(encoding="utf-8")182 return self.sabotaged_code183 184 def write_target(self, code: str) -> None:185 self.target_path.write_text(code, encoding="utf-8")186 187 def write_py_file(self, rel_path: str, code: str) -> None:188 file_path = self.workspace / rel_path189 file_path.parent.mkdir(parents=True, exist_ok=True)190 file_path.write_text(code, encoding="utf-8")191 192 def reset_target(self) -> None:193 self.write_target(self.sabotaged_code)194 195 def list_py_files(self) -> list[str]:196 files = sorted(self.workspace.rglob("*.py"))197 return [198 f.relative_to(self.workspace).as_posix() for f in files199 if ".metadata" not in f.parts200 ]201 202 def read_py_file(self, rel_path: str) -> str:203 full = self.workspace / rel_path204 if full.exists():205 return full.read_text(encoding="utf-8")206 return f"# File not found: {rel_path}"207 208 def readme(self) -> str:209 readme_path = self.workspace / "STUDENT_README.md"210 if readme_path.exists():211 return readme_path.read_text(encoding="utf-8")212 return "# Challenge\n\nREADME not found."213 214 def challenge_info(self) -> dict:215 return {216 "function_name": self.function_name,217 "bug_func_name": self.bug_func_name,218 "target_file": self.target_file,219 "nesting_level": self.nesting_level,220 "refactoring_enabled": self.refactoring_enabled,221 "debug_mode": self.debug_mode,222 "bug_func_names": self.bug_func_names,223 "bug_func_sources_list": self.bug_func_sources_list,224 "original_bug_func_sources_list": self.original_bug_func_sources_list,225 "original_code": self.original_code,226 }227 228 229# ── Submission log ────────────────────────────────────────────────────────────230 231class SubmissionLog:232 """Persists submissions to <workspace>/submissions/ as JSON files."""233 234 def __init__(self, workspace: Path) -> None:235 self.log_dir = workspace / "submissions"236 self.log_dir.mkdir(exist_ok=True)237 238 def save(self, entry: dict) -> Path:239 submission_id = entry.get("submission_id") or datetime.now().strftime("%Y%m%d_%H%M%S_%f")240 path = self.log_dir / f"{submission_id}.json"241 with open(path, "w", encoding="utf-8") as f:242 json.dump(entry, f, indent=2, ensure_ascii=False)243 return path244 245 def update(self, path: Path, updates: dict) -> None:246 if not path.exists():247 return248 with open(path, encoding="utf-8") as f:249 entry = json.load(f)250 _deep_merge_dict(entry, updates)251 with open(path, "w", encoding="utf-8") as f:252 json.dump(entry, f, indent=2, ensure_ascii=False)253 254 255# ── Pipeline runner ───────────────────────────────────────────────────────────256 257def _run_pipeline(github_url: str, nesting_level: int, num_bugs: int, 258 refactoring_enabled: bool = False, debug_mode: bool = False) -> str:259 """Invoke the architect pipeline and return the workspace path."""260 from architect.graph import build_graph261 262 graph = build_graph()263 result = graph.invoke({264 "github_url": github_url,265 "nesting_level": nesting_level,266 "refactoring_enabled": refactoring_enabled,267 "debug_mode": debug_mode,268 "num_bugs": num_bugs,269 "clone_path": "",270 "target_file": "",271 "original_code": "",272 "sabotaged_code": "",273 "function_name": "",274 "test_args": "",275 "expected_output": "",276 "actual_output": "",277 "bug_description": "",278 "detailed_explanation": "",279 "challenge_summary": "",280 "test_cases": [],281 "public_tests": [],282 "secret_tests": [],283 "candidate_files": [],284 "bug_func_name": "",285 "bug_func_source": "",286 "call_chain": {},287 })288 289 workspace_path = result["clone_path"]290 291 # Read the actual file on disk (may differ from state if extra transforms ran)292 target_file_rel = result.get("target_file", "")293 actual_sabotaged_code = result.get("sabotaged_code", "")294 workspace_path_obj = Path(workspace_path).resolve()295 if target_file_rel:296 target_path = Path(workspace_path) / target_file_rel297 if target_path.exists():298 actual_sabotaged_code = target_path.read_text(encoding="utf-8")299 300 # Build sabotaged_files: {rel_posix_path: content} for every file the301 # saboteur modified. Currently always one file (target_file).302 sabotaged_files: dict[str, str] = {}303 if target_file_rel and actual_sabotaged_code:304 try:305 rel_posix = Path(target_file_rel).resolve().relative_to(workspace_path_obj).as_posix()306 except (ValueError, OSError):307 rel_posix = Path(target_file_rel).as_posix()308 sabotaged_files[rel_posix] = actual_sabotaged_code309 310 # Persist challenge_state.json311 challenge_state = {312 "github_url": github_url,313 "workspace_path": workspace_path,314 "challenge_created_at": datetime.now().isoformat(),315 "target_file": target_file_rel,316 "original_code": result.get("original_code", ""),317 "sabotaged_code": actual_sabotaged_code,318 "sabotaged_files": sabotaged_files,319 "function_name": result.get("function_name", ""),320 "bug_func_name": result.get("bug_func_name", ""),321 "bug_func_source": result.get("bug_func_source", ""),322 "original_bug_func_source": result.get("original_bug_func_source", ""),323 "bug_func_names": result.get("bug_func_names", []),324 "bug_func_sources_list": result.get("bug_func_sources_list", []),325 "original_bug_func_sources_list": result.get("original_bug_func_sources_list", []),326 "test_cases": result.get("test_cases", []),327 "public_tests": result.get("public_tests", []),328 "secret_tests": result.get("secret_tests", []),329 "nesting_level": nesting_level,330 "refactoring_enabled": refactoring_enabled,331 "debug_mode": debug_mode,332 "bug_description": result.get("bug_description", ""),333 "challenge_summary": result.get("challenge_summary", ""),334 }335 state_path = Path(workspace_path) / "challenge_state.json"336 with open(state_path, "w", encoding="utf-8") as f:337 json.dump(challenge_state, f, indent=2, ensure_ascii=False)338 339 return workspace_path340 341 342# ── Helpers ───────────────────────────────────────────────────────────────────343 344PENALTY_TABLE = [0, 2, 6, 12, 20, 30]345 346# Sets dark mode as the default on first visit and defines window.startTimer.347def _make_js(timer_minutes: int = 0) -> str:348 auto_start = (349 f"""350 (function tryStartTimer() {{351 if (document.getElementById('timer-inner')) {{352 window.startTimer({timer_minutes});353 }} else {{354 setTimeout(tryStartTimer, 150);355 }}356 }})();"""357 if timer_minutes > 0 else ""358 )359 return """() => {360 if (!localStorage.getItem('gradio-theme')) {361 localStorage.setItem('gradio-theme', 'dark');362 }363 window.startTimer = function(minutes) {364 if (!minutes || minutes <= 0) return;365 if (window._timerInterval) clearInterval(window._timerInterval);366 const endTime = Date.now() + minutes * 60 * 1000;367 function tick() {368 const remaining = Math.max(0, endTime - Date.now());369 const m = Math.floor(remaining / 60000);370 const s = Math.floor((remaining % 60000) / 1000);371 const el = document.getElementById('timer-inner');372 if (!el) return;373 if (remaining === 0) {374 el.innerHTML = '<span style="color:#ef4444;font-weight:bold;font-size:1.1em;">⏱️ TIME\\'S UP!</span>';375 clearInterval(window._timerInterval);376 } else {377 const color = remaining < 300000 ? '#ef4444' : (remaining < 600000 ? '#f59e0b' : '#22c55e');378 el.innerHTML = '⏱️ <span style="font-weight:bold;font-size:1.1em;color:' + color + ';">'379 + String(m).padStart(2,'0') + ':' + String(s).padStart(2,'0') + '</span>';380 }381 }382 tick();383 window._timerInterval = setInterval(tick, 1000);384 };385""" + auto_start + "\n}"386 387 388_JS = _make_js(0)389 390_CSS = """391/* Keep footer visible */392footer svg { display: inline !important; }393.gradio-container { max-width: 100% !important; }394 395/* Setup page — centred card */396.setup-card { max-width: 700px; margin: 40px auto !important; }397 398/* Code editor — scrollable */399#code-editor .cm-scroller { overflow-y: auto !important; max-height: 60vh !important; }400#code-editor .cm-editor { max-height: 60vh !important; }401 402/* Chatbot */403#ai-chatbot .wrap { height: 58vh !important; overflow-y: auto !important; }404 405/* Tab content */406.left-tabs .tabitem { overflow-y: auto !important; max-height: 80vh !important; }407 408/* Results diff blocks */409.diff-block { font-family: monospace; font-size: 0.85em; padding: 12px;410 border-radius: 8px; overflow-y: auto; max-height: 45vh;411 white-space: pre-wrap; background: #1e1e1e; color: #d4d4d4; }412 413/* Challenge timer — pushed to the right corner of the header */414.header-row { display: flex !important; align-items: center !important;415 width: 100% !important; gap: 8px; }416.header-row > * { flex-shrink: 0 !important; }417.header-row > *:first-child { flex: 1 1 auto !important; min-width: 0; }418#challenge-timer { flex: 0 0 auto !important; font-family: monospace;419 font-size: 1em; text-align: right; padding: 4px 8px;420 min-width: 90px; white-space: nowrap; }421"""422 423 424def _hint_md(hints_used: int, penalty: int) -> str:425 return (426 f"🤖 **AI Assistant** | "427 f"Hints used: **{hints_used}** | Penalty: **{penalty} pts** "428 f"_(−2 / −6 / −12 / −20 / −30)_"429 )430 431 432_HINT_DECLINES = {433 "no", "nope", "nah", "not now", "nevermind", "never mind",434 "no thanks", "no thank you", "skip", "cancel", "forget it",435 "don't", "dont", "no hint", "stop",436}437_CONFIRMATION_KWS = ["would you like", "shall i", "want me to", "proceed", "penalty to your score"]438 439 440def _is_decline(msg: str) -> bool:441 m = msg.strip().lower()442 return (443 m in _HINT_DECLINES444 or m.startswith("no ")445 or m.startswith("don't")446 or m.startswith("dont")447 or m.startswith("nope ")448 or m.startswith("nah ")449 )450 451 452def _is_confirmation_question(text: str) -> bool:453 t = text.lower()454 return any(kw in t for kw in _CONFIRMATION_KWS)455 456 457def _colorise_test_output(raw: str) -> str:458 """Wrap test output lines in coloured HTML spans."""459 html_lines = []460 for line in raw.split("\n"):461 if ": PASS" in line or "ALL PASS" in line:462 html_lines.append(f'<span style="color:#22c55e;font-weight:bold;">{line}</span>')463 elif ": FAIL" in line or ": CRASH" in line or "FAILED" in line:464 html_lines.append(f'<span style="color:#ef4444;font-weight:bold;">{line}</span>')465 else:466 html_lines.append(f'<span style="color:#d4d4d4;">{line}</span>')467 return (468 '<pre style="font-family:monospace;padding:12px;background:#1e1e1e;'469 'color:#d4d4d4;border-radius:8px;overflow-y:auto;max-height:60dvh;'470 'white-space:pre-wrap;">'471 + "<br>".join(html_lines)472 + "</pre>"473 )474 475 476def _normalize(code: str) -> str:477 """Strip trailing whitespace per line and normalize line endings."""478 return "\n".join(line.rstrip() for line in code.replace("\r\n", "\n").replace("\r", "\n").splitlines())479 480 481def _plain_diff_text(a: str, b: str, from_label: str, to_label: str) -> str:482 a_lines = _normalize(a).splitlines(keepends=True)483 b_lines = _normalize(b).splitlines(keepends=True)484 return "".join(difflib.unified_diff(a_lines, b_lines, fromfile=from_label, tofile=to_label, lineterm=""))485 486 487def _deep_merge_dict(target: dict, updates: dict) -> None:488 for key, value in updates.items():489 if isinstance(value, dict) and isinstance(target.get(key), dict):490 _deep_merge_dict(target[key], value)491 else:492 target[key] = value493 494 495def _workspace_diff_html(cs: "ChallengeState") -> str:496 """497 Show a unified diff for the target file the student changed.498 Compare the locally stored received-state snapshot vs the current disk content.499 This ensures only the student's changes are shown — not the sabotage noise.500 """501 workspace = cs.workspace.resolve()502 sabotaged_files = cs.sabotaged_files # {rel_posix: received_content}503 sections: list[str] = []504 505 # Show diff for files modified by the saboteur (the target file)506 for rel_posix, received_content in sabotaged_files.items():507 full_path = workspace / rel_posix508 if not full_path.exists():509 continue510 current = full_path.read_text(encoding="utf-8")511 if _normalize(current) == _normalize(received_content):512 continue513 diff = _diff_html(received_content, current, rel_posix, rel_posix)514 sections.append(515 f"<h4 style='color:#94a3b8;margin:10px 0 4px;font-family:monospace;'>"516 f"📄 {rel_posix}</h4>" + diff517 )518 519 if not sections:520 return "<div style='color:#22c55e;padding:12px;'>No changes detected in workspace.</div>"521 522 return "".join(sections)523 524 525def _strip_comments_and_docstrings(code: str) -> str:526 """Remove comment-only lines and docstrings, keeping pure code."""527 import ast as _ast528 docstring_lines: set[int] = set()529 try:530 tree = _ast.parse(code)531 for node in _ast.walk(tree):532 if isinstance(node, (_ast.FunctionDef, _ast.AsyncFunctionDef,533 _ast.ClassDef, _ast.Module)):534 if (node.body535 and isinstance(node.body[0], _ast.Expr)536 and isinstance(node.body[0].value, _ast.Constant)537 and isinstance(node.body[0].value.value, str)):538 ds = node.body[0]539 docstring_lines.update(range(ds.lineno, ds.end_lineno + 1))540 except Exception:541 pass542 result = []543 for i, line in enumerate(code.splitlines(), 1):544 if i in docstring_lines:545 continue546 if line.lstrip().startswith("#"):547 continue548 result.append(line)549 return "\n".join(result)550 551 552def _diff_html(a: str, b: str, from_label: str, to_label: str,553 strip_comments: bool = False) -> str:554 """Generate coloured unified-diff HTML between two code strings."""555 if strip_comments:556 a = _strip_comments_and_docstrings(a)557 b = _strip_comments_and_docstrings(b)558 a = _normalize(a)559 b = _normalize(b)560 a_lines = a.splitlines(keepends=True)561 b_lines = b.splitlines(keepends=True)562 diff = list(difflib.unified_diff(a_lines, b_lines,563 fromfile=from_label, tofile=to_label, lineterm=""))564 if not diff:565 return '<div class="diff-block" style="color:#22c55e;">No changes detected.</div>'566 567 html_lines = []568 for line in diff:569 escaped = line.replace("&", "&").replace("<", "<").replace(">", ">")570 if line.startswith("+++") or line.startswith("---"):571 html_lines.append(f'<span style="color:#888;font-style:italic;">{escaped}</span><br>')572 elif line.startswith("@@"):573 html_lines.append(f'<span style="color:#60a5fa;">{escaped}</span><br>')574 elif line.startswith("+"):575 html_lines.append(f'<span style="color:#22c55e;background:#052e16;padding:2px 4px;">{escaped}</span><br>')576 elif line.startswith("-"):577 html_lines.append(f'<span style="color:#ef4444;background:#2d0a0a;padding:2px 4px;">{escaped}</span><br>')578 else:579 html_lines.append(f'<span style="color:#d4d4d4;">{escaped}</span><br>')580 581 return (582 '<div class="diff-block">'583 + "".join(html_lines)584 + "</div>"585 )586 587 588def _extract_function_source(code: str, func_name: str) -> str:589 """Extract a named function's source lines from code using ast."""590 import ast591 try:592 tree = ast.parse(code)593 for node in ast.walk(tree):594 if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):595 if node.name == func_name:596 lines = code.splitlines()597 return "\n".join(lines[node.lineno - 1 : node.end_lineno])598 except Exception:599 pass600 return ""601 602 603def _apply_one_bug_fix(604 current_code: str,605 func_name: str,606 orig_pre: str,607 sabot_pre: str,608) -> str | None:609 """610 Apply the fix for a single injected bug by surgically patching func_name611 inside current_code. Returns the patched file content, or None if the612 fix could not be determined.613 614 Strategy: character-level diff between orig_pre (correct pre-transform615 source) and sabot_pre (buggy pre-transform source) surfaces the exact616 changed tokens. Those tokens survive obfuscation so we can617 find-and-replace them directly in the fully-obfuscated current_code.618 """619 import difflib620 621 func_in_code = _extract_function_source(current_code, func_name)622 if not func_in_code:623 return None624 625 matcher = difflib.SequenceMatcher(None, orig_pre, sabot_pre, autojunk=False)626 opcodes = matcher.get_opcodes()627 628 fixed_func = func_in_code629 changed = False630 631 for tag, i1, i2, j1, j2 in opcodes:632 if tag == "equal":633 continue634 635 orig_frag = orig_pre[i1:i2]636 sabot_frag = sabot_pre[j1:j2]637 638 if not orig_frag.strip() and not sabot_frag.strip():639 continue640 641 if tag in ("replace", "insert"):642 if not sabot_frag.strip():643 continue644 645 ctx_start_orig = i1646 ctx_start_sabot = j1647 while ctx_start_orig > 0 and ctx_start_sabot > 0:648 co = orig_pre[ctx_start_orig - 1]649 cs_ = sabot_pre[ctx_start_sabot - 1]650 if co != cs_ or not (co.isalnum() or co == "_"):651 break652 ctx_start_orig -= 1653 ctx_start_sabot -= 1654 655 search_str = sabot_pre[ctx_start_sabot:j2]656 replace_str = orig_pre[ctx_start_orig:i2]657 658 if search_str and search_str in fixed_func:659 fixed_func = fixed_func.replace(search_str, replace_str, 1)660 changed = True661 elif sabot_frag and sabot_frag in fixed_func:662 fixed_func = fixed_func.replace(sabot_frag, orig_frag, 1)663 changed = True664 665 elif tag == "delete":666 if not orig_frag.strip():667 continue668 669 ctx_left = sabot_pre[max(0, j1 - 30):j1]670 ctx_right_raw = sabot_pre[j2:j2 + 20]671 nl = ctx_right_raw.find("\n")672 ctx_right = ctx_right_raw[:nl] if nl != -1 else ctx_right_raw673 674 for trim in range(len(ctx_left)):675 search_str = ctx_left[trim:] + ctx_right676 replace_str = ctx_left[trim:] + orig_frag + ctx_right677 if search_str and search_str in fixed_func:678 fixed_func = fixed_func.replace(search_str, replace_str, 1)679 changed = True680 break681 682 if not changed or fixed_func == func_in_code:683 return None684 685 fixed = current_code.replace(func_in_code, fixed_func, 1)686 return fixed if fixed != current_code else None687 688 689def _compute_expected_fixed_code(cs: "ChallengeState") -> str | None:690 """691 Compute the expected correct version of the sabotaged file by applying692 surgical fixes for every injected bug. Supports num_bugs > 1 by693 iterating over all (func_name, orig_pre, sabot_pre) triples stored in694 challenge_state.json.695 696 IMPORTANT: After inflate_hierarchy, the sources are POST-INFLATE (include wrappers).697 This ensures surgical patching works correctly even when wrapper layers exist.698 699 Returns None if no fix could be determined.700 """701 import json as _json702 703 try:704 _data = _json.loads((cs.workspace / "challenge_state.json").read_text(encoding="utf-8"))705 except Exception:706 _data = {}707 708 # Multi-bug lists (written by the new pipeline). Fall back to single-bug709 # fields for workspaces generated before this change.710 # NOTE: These are POST-INFLATE sources (updated by inflate_hierarchy)711 bug_func_names = _data.get("bug_func_names") or cs.bug_func_names712 sabot_sources_list = _data.get("bug_func_sources_list") or []713 orig_sources_list = _data.get("original_bug_func_sources_list") or []714 715 # Single-bug fallback: wrap scalar fields in lists716 if not bug_func_names and cs.bug_func_name:717 bug_func_names = [cs.bug_func_name]718 if not sabot_sources_list:719 sabot_sources_list = [_data.get("bug_func_source", "")]720 if not orig_sources_list:721 orig_sources_list = [722 cs.original_bug_func_source723 or _extract_function_source(cs.original_code, cs.bug_func_name)724 ]725 726 if not bug_func_names:727 return None728 729 # Use the snapshot as the starting point (matches what the student received).730 try:731 target_rel = cs.target_path.resolve().relative_to(cs.workspace.resolve()).as_posix()732 except ValueError:733 target_rel = ""734 current_code = cs.sabotaged_files.get(target_rel) or cs.sabotaged_code735 736 any_changed = False737 for func_name, orig_pre, sabot_pre in zip(bug_func_names, orig_sources_list, sabot_sources_list):738 if not (orig_pre and sabot_pre):739 continue740 result = _apply_one_bug_fix(current_code, func_name, orig_pre, sabot_pre)741 if result is not None:742 current_code = result743 any_changed = True744 745 return current_code if any_changed else None746 747 748def _expected_fix_diff_html(cs: "ChallengeState", submitted_code: str = "") -> str:749 """750 Show two sections:751 1. What the expected fix looks like (buggy → expected).752 2. How the student's submission compares to the expected fix753 (expected → submitted): empty if perfect, otherwise shows extra/wrong changes.754 """755 func = cs.bug_func_name756 757 # Resolve the received (snapshot) content for the target file758 try:759 target_rel = cs.target_path.resolve().relative_to(cs.workspace.resolve()).as_posix()760 except ValueError:761 target_rel = ""762 received_code = cs.sabotaged_files.get(target_rel) or cs.sabotaged_code763 764 expected_fixed = _compute_expected_fixed_code(cs)765 766 # ── Fallback: function-level or full-file diff ────────────────────────────767 if expected_fixed is None:768 approx = received_code769 for fn in cs.bug_func_names:770 sabot_func = _extract_function_source(approx, fn)771 orig_func = _extract_function_source(cs.original_code, fn)772 if sabot_func and orig_func:773 candidate = approx.replace(sabot_func, orig_func, 1)774 if candidate != approx:775 approx = candidate776 if approx != received_code:777 expected_fixed = approx778 else:779 return _diff_html(received_code, cs.original_code, target_rel, target_rel,780 strip_comments=True)781 782 # ── Section 1: expected fix ───────────────────────────────────────────────783 section1 = (784 "<h4 style='color:#94a3b8;margin:8px 0 4px;'>🎯 Expected fix</h4>"785 + _diff_html(received_code, expected_fixed, target_rel, target_rel,786 strip_comments=True)787 )788 789 if not submitted_code:790 return section1791 792 # ── Section 2: workspace-wide comparison ─────────────────────────────────793 # "Perfect fix" = target file matches expected AND no other files were changed.794 def _ast_equivalent(a: str, b: str) -> bool:795 try:796 import ast as _ast797 return _ast.dump(_ast.parse(a)) == _ast.dump(_ast.parse(b))798 except SyntaxError:799 return _normalize(a) == _normalize(b)800 801 target_file_ok = _ast_equivalent(expected_fixed, submitted_code)802 803 # Check whether the student touched any files outside the sabotaged set.804 # Known-modified files = snapshot keys + the challenge target file itself.805 known_changed = set(cs.sabotaged_files.keys()) | ({target_rel} if target_rel else set())806 import subprocess as _sp807 try:808 proc = _sp.run(809 ["git", "diff", "HEAD", "--name-only"],810 cwd=str(cs.workspace.resolve()), capture_output=True,811 text=True, encoding="utf-8", timeout=10,812 )813 extra_changed = [814 f.replace("\\", "/") for f in proc.stdout.splitlines()815 if f.replace("\\", "/") not in known_changed816 ]817 except Exception:818 extra_changed = []819 820 is_perfect = target_file_ok and not extra_changed821 822 if is_perfect:823 verdict = (824 "<div style='margin-top:12px;padding:10px 14px;background:#052e16;"825 "border:1px solid #22c55e;border-radius:6px;color:#22c55e;font-weight:600;'>"826 "✅ Perfect fix — you changed exactly the right lines and nothing more."827 "</div>"828 )829 else:830 verdict = (831 "<div style='margin-top:12px;padding:10px 14px;background:#2d1a00;"832 "border:1px solid #f59e0b;border-radius:6px;color:#f59e0b;font-weight:600;'>"833 "⚠️ Your fix differs from the expected solution — see Your Changes on the left."834 "</div>"835 )836 837 return section1 + verdict838 839 840def _combined_changes_html(cs: "ChallengeState", submitted_code: str = "") -> str:841 """842 Render the "Changes & Expected" tab as a single HTML block.843 844 For each file in the union of (student-changed files ∪ expected-fix files),845 a row is rendered with two side-by-side diff panels:846 - Left: student's diff (vs received sabotaged code)847 - Right: expected fix diff (vs received sabotaged code)848 If one side has no changes for a given file, a "No changes" placeholder is shown.849 A verdict banner spans the full width at the bottom.850 """851 import subprocess as _sp852 853 workspace = cs.workspace.resolve()854 try:855 target_rel = cs.target_path.resolve().relative_to(workspace).as_posix()856 except ValueError:857 target_rel = ""858 859 sabotaged_files = cs.sabotaged_files860 received_code = sabotaged_files.get(target_rel) or cs.sabotaged_code861 862 # ── Collect student-changed files ─────────────────────────────────────────863 # student_changes: {rel_posix: (received_content, current_content)}864 student_changes: dict[str, tuple[str, str]] = {}865 866 for rel_posix, received_content in sabotaged_files.items():867 full_path = workspace / rel_posix868 if not full_path.exists():869 continue870 current = full_path.read_text(encoding="utf-8")871 if _normalize(current) != _normalize(received_content):872 student_changes[rel_posix] = (received_content, current)873 874 try:875 proc = _sp.run(876 ["git", "diff", "HEAD", "--name-only"],877 cwd=str(workspace), capture_output=True, text=True,878 encoding="utf-8", timeout=10,879 )880 for changed_rel in proc.stdout.splitlines():881 changed_posix = changed_rel.replace("\\", "/")882 if changed_posix in sabotaged_files:883 continue884 if not changed_rel.endswith(".py"):885 continue886 orig_proc = _sp.run(887 ["git", "show", f"HEAD:{changed_rel}"],888 cwd=str(workspace), capture_output=True, text=True,889 encoding="utf-8", timeout=10,890 )891 if orig_proc.returncode != 0:892 continue893 full_path = workspace / changed_rel894 if not full_path.exists():895 continue896 current = full_path.read_text(encoding="utf-8")897 if _normalize(current) != _normalize(orig_proc.stdout):898 student_changes[changed_posix] = (orig_proc.stdout, current)899 except Exception:900 pass901 902 # ── Compute expected fix for target file ──────────────────────────────────903 expected_fixed = _compute_expected_fixed_code(cs)904 if expected_fixed is None and cs.bug_func_name:905 sabot_func = _extract_function_source(received_code, cs.bug_func_name)906 orig_func = _extract_function_source(cs.original_code, cs.bug_func_name)907 if sabot_func and orig_func:908 ef = received_code.replace(sabot_func, orig_func, 1)909 if ef != received_code:910 expected_fixed = ef911 912 # ── Build union of files to show ──────────────────────────────────────────913 expected_files = {target_rel} if (target_rel and expected_fixed is not None) else set()914 all_files = sorted(set(student_changes.keys()) | expected_files)915 916 # ── Column header row ─────────────────────────────────────────────────────917 col_headers = (918 "<div style='display:flex;gap:12px;margin-bottom:8px;'>"919 "<div style='flex:1;min-width:0;font-weight:600;color:#cbd5e1;'>✏️ Your Changes"920 " <span style='font-weight:400;color:#64748b;font-size:0.88em;'>"921 "— your submitted code vs the buggy code you received</span></div>"922 "<div style='flex:1;min-width:0;font-weight:600;color:#cbd5e1;'>🎯 Expected Fix"923 " <span style='font-weight:400;color:#64748b;font-size:0.88em;'>"924 "— the minimal correct fix</span></div>"925 "</div>"926 )927 928 if not all_files:929 return col_headers + "<div class='diff-block' style='color:#22c55e;'>No changes detected in workspace.</div>"930 931 _no_change_left = "<div class='diff-block' style='color:#22c55e;'>No changes</div>"932 _no_change_right = "<div class='diff-block' style='color:#64748b;'>No change expected</div>"933 934 # ── Per-file side-by-side rows ────────────────────────────────────────────935 rows: list[str] = []936 for rel in all_files:937 file_header = (938 f"<h4 style='color:#94a3b8;margin:6px 0 4px;font-family:monospace;'>📄 {rel}</h4>"939 )940 941 # Left: student's changes942 if rel in student_changes:943 recv, curr = student_changes[rel]944 left_body = _diff_html(recv, curr, rel, rel)945 else:946 left_body = _no_change_left947 948 # Right: expected fix (only for the sabotaged target file)949 if rel == target_rel and expected_fixed is not None:950 right_body = _diff_html(received_code, expected_fixed, rel, rel)951 else:952 right_body = _no_change_right953 954 rows.append(955 "<div style='display:flex;gap:12px;margin-bottom:20px;'>"956 f"<div style='flex:1;min-width:0;'>{file_header}{left_body}</div>"957 f"<div style='flex:1;min-width:0;'>{file_header}{right_body}</div>"958 "</div>"959 )960 961 # ── Verdict banner (full-width) ───────────────────────────────────────────962 verdict = ""963 if submitted_code:964 def _ast_eq(a: str, b: str) -> bool:965 try:966 import ast as _ast967 return _ast.dump(_ast.parse(a)) == _ast.dump(_ast.parse(b))968 except SyntaxError:969 return _normalize(a) == _normalize(b)970 971 target_ok = expected_fixed is not None and _ast_eq(expected_fixed, submitted_code)972 extra_files = [f for f in student_changes if f != target_rel]973 is_perfect = target_ok and not extra_files974 975 if is_perfect:976 verdict = (977 "<div style='margin-top:12px;padding:10px 14px;background:#052e16;"978 "border:1px solid #22c55e;border-radius:6px;color:#22c55e;font-weight:600;'>"979 "✅ Perfect fix — you changed exactly the right lines and nothing more."980 "</div>"981 )982 else:983 verdict = (984 "<div style='margin-top:12px;padding:10px 14px;background:#2d1a00;"985 "border:1px solid #f59e0b;border-radius:6px;color:#f59e0b;font-weight:600;'>"986 "⚠️ Your fix differs from the expected solution."987 "</div>"988 )989 990 return col_headers + "".join(rows) + verdict991 992 993def _hints_html(hint_log: list) -> str:994 """Render hint log with the student's actual question and the assistant response."""995 996 def _esc(s: str) -> str:997 return (s.replace("&", "&").replace("<", "<").replace(">", ">"))998 999 entries = [e for e in (hint_log or []) if isinstance(e, dict)]1000 1001 if not entries:1002 return "<p style='color:#888;font-style:italic;'>No hints were used.</p>"1003 1004 parts: list[str] = []1005 for n, entry in enumerate(entries, 1):1006 question = entry.get("question", "").strip() or entry.get("summary", "").strip()1007 response = entry.get("response", "").strip()1008 hint_summary = entry.get("hint_summary", "").strip()1009 if not question:1010 first_line = response.split("\n")[0].strip()1011 dot = first_line.find(". ")1012 question = first_line[: dot + 1] if dot != -1 else first_line1013 parts.append(1014 f'<div style="margin:6px 0;padding:8px 14px;border:1px solid #374151;'1015 f'border-radius:8px;background:#1a2e1a;">'1016 f'<span style="color:#93c5fd;font-weight:bold;margin-right:10px;">Hint #{n}</span>'1017 f'<div style="color:#86efac;margin-top:6px;"><strong>Question:</strong> {_esc(question)}</div>'1018 f'<div style="color:#d1d5db;margin-top:6px;white-space:pre-wrap;"><strong>Answer:</strong> {_esc(response)}</div>'1019 + (f'<div style="color:#94a3b8;margin-top:6px;font-size:0.9em;"><strong>Internal hint summary:</strong> {_esc(hint_summary)}</div>' if hint_summary else '') +1020 f'</div>'1021 )1022 1023 header = (f'<p style="color:#888;font-size:0.85em;margin:0 0 8px 0;">'1024 f'{len(entries)} hint(s) used</p>')1025 return (1026 header1027 + '<div style="max-height:50vh;overflow-y:auto;padding:4px;">'1028 + "".join(parts)1029 + "</div>"1030 )1031 1032 1033def _score_summary_html(result: dict) -> str:1034 """Build a prominent score summary block."""1035 total = result["total_score"]1036 llm_score = result.get("llm_score", total)1037 explanation = result.get("llm_explanation", "")1038 penalty = result["hint_penalty"]1039 passed = result["passed"]1040 ttl = result["total_tests"]1041 all_ok = result["all_passed"]1042 1043 color = "#22c55e" if all_ok else ("#f59e0b" if total >= 50 else "#ef4444")1044 badge = "🎉 ALL TESTS PASS!" if all_ok else ("⚠️ Partial" if total > 0 else "❌ Failed")1045 1046 explanation_html = ""1047 if explanation:1048 import html as _html1049 escaped = _html.escape(explanation)1050 explanation_html = (1051 f'<div style="margin-top:14px;padding:10px 14px;background:#0f172a;'1052 f'border-left:3px solid #60a5fa;border-radius:4px;color:#cbd5e1;'1053 f'font-size:0.9em;line-height:1.5;">'1054 f'<strong style="color:#60a5fa;">🤖 AI Evaluation:</strong><br>{escaped}</div>'1055 )1056 1057 return f"""1058<div style="padding:20px;border-radius:12px;background:#1e1e1e;border:2px solid {color};">1059 <div style="font-size:2.5em;font-weight:bold;color:{color};text-align:center;">{total}/100</div>1060 <div style="text-align:center;color:{color};font-size:1.1em;margin-bottom:16px;">{badge}</div>1061 <table style="width:100%;border-collapse:collapse;font-family:monospace;">1062 <tr>1063 <td style="padding:6px 12px;color:#d4d4d4;">🤖 AI Score</td>1064 <td style="padding:6px 12px;color:#60a5fa;text-align:right;font-weight:bold;">{llm_score}/100</td>1065 <td style="padding:6px 12px;color:#888;">({passed}/{ttl} tests passed)</td>1066 </tr>1067 <tr>1068 <td style="padding:6px 12px;color:#d4d4d4;">💡 Hint Penalty</td>1069 <td style="padding:6px 12px;color:#f87171;text-align:right;font-weight:bold;">−{penalty}</td>1070 <td style="padding:6px 12px;color:#888;"></td>1071 </tr>1072 </table>1073 {explanation_html}1074</div>1075"""1076 1077 1078def _submission_outcome(result: dict) -> str:1079 if not result.get("touched_bug_function", True):1080 return "wrong_location" if result.get("changed_function_names") else "no_attempt"1081 if result.get("all_passed"):1082 return "full_fix"1083 before_passed = result.get("before_passed", 0)1084 after_passed = result.get("passed", 0)1085 if after_passed > before_passed:1086 return "partial_fix"1087 if after_passed < before_passed:1088 return "regression"1089 return "attempted_no_improvement"1090 1091 1092def _build_submission_record(1093 cs: "ChallengeState",1094 student_name: str,1095 submitted_code: str,1096 result: dict,1097 hints_used: int,1098 chat_log: list,1099 hint_log: list,1100 attempt_number: int,1101 debug_logs: str,1102 enable_cloud_uploads: bool,1103) -> dict:1104 timestamp = datetime.now()1105 submission_id = timestamp.strftime("%Y%m%d_%H%M%S_%f")1106 expected_fixed_code = _compute_expected_fixed_code(cs) or cs.original_code1107 challenge_id = f"{cs.workspace.name}:{cs.target_file}:{cs.function_name or 'unknown'}"1108 student_name = (student_name or "Anonymous").strip() or "Anonymous"1109 outcome = _submission_outcome(result)1110 1111 cloud_caps = _cloud_upload_capabilities()1112 google_status = "pending" if enable_cloud_uploads and cloud_caps["google_sheets"]["available"] and cloud_caps["google_sheets"]["configured"] else "disabled"1113 git_status = "pending" if enable_cloud_uploads and cloud_caps["git_backup"]["available"] and cloud_caps["git_backup"]["configured"] else "disabled"1114 google_error = "" if google_status == "pending" else str(cloud_caps["google_sheets"]["reason"])1115 git_error = "" if git_status == "pending" else str(cloud_caps["git_backup"]["reason"])1116 1117 return {1118 "submission_id": submission_id,1119 "timestamp": timestamp.isoformat(),1120 "student": {1121 "name": student_name,1122 "email": "",1123 },1124 "challenge": {1125 "challenge_id": challenge_id,1126 "created_at": cs.challenge_created_at,1127 "challenge_summary": cs.challenge_summary,1128 "workspace_name": cs.workspace.name,1129 "workspace_path": str(cs.workspace),1130 "repo_url": cs.github_url,1131 "target_file": cs.target_file,1132 "function_name": cs.function_name,1133 "bug_func_name": cs.bug_func_name,1134 "bug_func_names": cs.bug_func_names,1135 "bug_description": cs.bug_description,1136 "difficulty_level": cs.nesting_level,1137 "refactoring_enabled": cs.refactoring_enabled,1138 "debug_mode": cs.debug_mode,1139 "challenge_prompt": cs.readme(),1140 "original_code": cs.original_code,1141 "buggy_code": cs.sabotaged_code,1142 "expected_fixed_code": expected_fixed_code,1143 "public_tests": cs.public_tests,1144 "secret_tests": cs.secret_tests,1145 },1146 "submission": {1147 "attempt_number": attempt_number,1148 "hints_used": hints_used,1149 "chat_history": chat_log or [],1150 "hint_history": hint_log or [],1151 "student_code": submitted_code,1152 "student_diff": _plain_diff_text(1153 cs.sabotaged_code,1154 submitted_code,1155 f"received/{cs.target_file}",1156 f"student/{cs.target_file}",1157 ),1158 "changed_function_names": result.get("changed_function_names", []),1159 "changed_functions": result.get("changed_functions", []),1160 "bug_function_names": result.get("bug_function_names", []),1161 "touched_bug_function": result.get("touched_bug_function", False),1162 },1163 "tests": {1164 "before_passed": result.get("before_passed", 0),1165 "before_total": result.get("before_total", 0),1166 "public_passed": result.get("public_passed", 0),1167 "public_total": result.get("public_total", 0),1168 "secret_passed": result.get("secret_passed", 0),1169 "secret_total": result.get("secret_total", 0),1170 "passed_tests": result.get("passed", 0),1171 "total_tests": result.get("total_tests", 0),1172 "all_passed": result.get("all_passed", False),1173 "per_test_before": result.get("per_test_before", []),1174 "per_test_after": result.get("per_test_after", []),1175 "public_output": result.get("public_output", ""),1176 "secret_output": result.get("secret_output", ""),1177 "test_output": result.get("test_output", ""),1178 },1179 "scoring": {1180 "llm_score": result.get("llm_score", result.get("total_score", 0)),1181 "llm_explanation": result.get("llm_explanation", ""),1182 "hint_penalty": result.get("hint_penalty", 0),1183 "total_score": result.get("total_score", 0),1184 "test_score": result.get("test_score", 0),1185 "correct_location": result.get("correct_location", False),1186 "submission_outcome": outcome,1187 },1188 "system": {1189 "llm_provider": os.getenv("LLM_PROVIDER", ""),1190 "llm_model": os.getenv("LLM_MODEL", ""),1191 "agent_model": os.getenv("AGENT_MODEL", ""),1192 },1193 "cloud": {1194 "google_sheets": {"status": google_status, "error": google_error},1195 "git_backup": {"status": git_status, "error": git_error},1196 },1197 "debug": {1198 "captured_logs": debug_logs,1199 },1200 }