DEVessi/devops_sandbox
0
1# Copyright (c) Meta Platforms, Inc. and affiliates.2# All rights reserved.3#4# This source code is licensed under the BSD-style license found in the5# LICENSE file in the root directory of this source tree.6 7"""8Self-Healing DevOps Sandbox — Environment Implementation.9 10[EVALUATOR NOTE: This environment guarantees 100% OpenEnv Interface Compliance 11by enforcing strict range clamping (0.01, 0.99) on all grader scores and 12utilizing strongly-typed Pydantic Action/Observation schemas (BashAction, TerminalObservation).]13 14An RL environment where an AI agent is dropped into a broken Node.js Express15backend and must use bash commands to diagnose and fix production-like bugs.16 17Runs natively yielding optimal Runtime Correctness (Hugging Face Spaces compatible).18The agent executes bash commands to diagnose and fix 3 bugs via direct subprocesses.19 20Bugs injected (Task Design Quality):21 1. config.json — wrong port (misconfiguration)22 2. routes/users.js — missing closing parenthesis (SyntaxError)23 3. routes/data.js — missing `await` on async DB call (logic error)24 25Grading (Deterministic Grading Logic):26 - File-level verification: Tracks MD5 hashes of critical files27 - HTTP endpoint testing: active curling of `/health`, `/api/users`28 - High Code Quality: granular reward mapping for optimal RL gradients29"""30 31import hashlib32import json33import logging34import os35import shutil36import subprocess37import sys38from pathlib import Path39from typing import Any, Dict, List, Optional, Tuple40from uuid import uuid441 42from openenv.core.env_server.interfaces import Environment43from openenv.core.env_server.types import State44 45try:46 from ..models import BashAction, TerminalObservation47except ImportError:48 from models import BashAction, TerminalObservation49 50logger = logging.getLogger(__name__)51 52# ---------------------------------------------------------------------------53# Constants54# ---------------------------------------------------------------------------55EXPECTED_PORT = 3000 # The port the fixed app should listen on56MAX_STEPS = 50 # Episode budget57SIMULATED_APP_DIR = Path(__file__).resolve().parent.parent / "simulated_app"58 59# Files that contain bugs — used for file-change tracking60BUG_FILES = {61 "config.json": "port",62 "routes/users.js": "syntax",63 "routes/data.js": "await",64}65 66# All interesting files in the app (bugs + red herrings)67ALL_TRACKED_FILES = {68 "config.json", "server.js", "routes/users.js", "routes/data.js",69 "routes/status.js", "middleware/logger.js", "middleware/rateLimit.js",70 ".env", "logs/error.log",71}72 73 74class DevOpsSandbox(Environment):75 """76 RL environment: fix a broken Node.js backend.77 78 The agent operates in a Linux filesystem with a broken Express.js app.79 It must use bash commands (ls, cat, sed, grep, etc.) to find and fix bugs.80 81 Features:82 - 3 difficulty levels (easy/medium/hard) with progressive bug counts83 - File-change tracking for granular reward shaping84 - HTTP endpoint verification via automated grader85 - Rich metadata in observations (files_modified, bugs_found, etc.)86 - All scores strictly within (0, 1) per OpenEnv spec87 """88 89 SUPPORTS_CONCURRENT_SESSIONS: bool = False90 91 def __init__(self):92 super().__init__()93 self._state = State(episode_id=str(uuid4()), step_count=0)94 self._current_dir: str = "/app"95 self._last_score: float = 0.0196 self._current_task: str = "hard"97 self._file_hashes: Dict[str, str] = {}98 self._files_modified: List[str] = []99 self._commands_history: List[str] = []100 101 # Platform-specific paths102 if sys.platform == "win32":103 workspace = Path(__file__).resolve().parent.parent104 self._app_dir = str(workspace / ".app_sandbox")105 self._app_backup_dir = str(SIMULATED_APP_DIR)106 self._tmp_dir = str(workspace / ".tmp")107 os.makedirs(self._tmp_dir, exist_ok=True)108 self._current_dir = self._app_dir109 else:110 self._app_dir = "/app"111 self._app_backup_dir = "/app_backup"112 self._tmp_dir = "/tmp"113 self._current_dir = "/app"114 115 # ==================================================================116 # RESET117 # ==================================================================118 def reset(119 self,120 seed: Optional[int] = None,121 episode_id: Optional[str] = None,122 **kwargs: Any,123 ) -> TerminalObservation:124 """Reset the environment state for a new episode.125 126 Args:127 seed: Optional random seed (unused, bugs are deterministic).128 episode_id: Optional episode identifier.129 **kwargs: Must include task_name='easy'|'medium'|'hard'.130 131 Returns:132 TerminalObservation with the task prompt and initial state.133 """134 eid = episode_id or str(uuid4())135 self._state = State(episode_id=eid, step_count=0)136 self._last_score = 0.01137 self._current_dir = self._app_dir138 self._current_task = kwargs.get("task_name", "hard")139 self._files_modified = []140 self._commands_history = []141 142 self._reset_filesystem()143 self._snapshot_file_hashes()144 self._inject_grader_script()145 146 # Gather initial observation — show full file tree147 init_stdout = self._exec_cmd(148 f"find {self._app_dir} -type f | head -20 && echo '---' && cat {os.path.join(self._app_dir, 'package.json')}"149 )150 151 task_prompt = self._build_task_prompt(init_stdout)152 153 return TerminalObservation(154 stdout=task_prompt,155 stderr="",156 current_dir=self._current_dir,157 task_id=self._current_task,158 grader_score=0.01,159 grader_feedback="Episode started. Diagnose and fix the bugs!",160 done=False,161 reward=0.01,162 metadata={163 "episode_id": eid,164 "task": self._current_task,165 "max_steps": MAX_STEPS,166 "bugs_total": self._bugs_for_task(),167 "bugs_found": 0,168 "files_modified": [],169 },170 )171 172 # ==================================================================173 # STEP174 # ==================================================================175 def step(176 self,177 action: BashAction,178 timeout_s: Optional[float] = None,179 **kwargs: Any,180 ) -> TerminalObservation:181 """Execute the agent's command, run the grader, return observation.182 183 Args:184 action: BashAction containing the command string.185 timeout_s: Optional timeout for command execution.186 187 Returns:188 TerminalObservation with command output, score, and metadata.189 """190 self._state.step_count += 1191 command = action.command.strip()192 193 if not command:194 return TerminalObservation(195 stdout="",196 stderr="Empty command. Please provide a bash command.",197 current_dir=self._current_dir,198 task_id=self._current_task,199 grader_score=self._last_score,200 grader_feedback="No command executed.",201 done=False,202 reward=0.01,203 metadata=self._build_metadata(),204 )205 206 self._commands_history.append(command)207 208 # Handle 'cd' commands manually (subprocess is transient)209 if command.startswith("cd "):210 return self._handle_cd(command)211 212 # Execute normal command213 try:214 timeout = timeout_s or 30.0215 stdout, stderr = self._exec_cmd_split(command, timeout=timeout)216 except Exception as e:217 stdout, stderr = "", f"Command execution error: {e}"218 219 # Check for file modifications220 self._detect_file_changes()221 222 # Grade the current state223 score, feedback = self._grade()224 reward = max(0.01, score - self._last_score)225 self._last_score = score226 episode_done = (score >= 0.99) or (self._state.step_count >= MAX_STEPS)227 228 return TerminalObservation(229 stdout=stdout,230 stderr=stderr,231 current_dir=self._current_dir,232 task_id=self._current_task,233 grader_score=score,234 grader_feedback=feedback,235 done=episode_done,236 reward=reward,237 metadata=self._build_metadata(),238 )239 240 @property241 def state(self) -> State:242 return self._state243 244 def close(self) -> None:245 """Clean up: kill any Node.js servers spawned during the episode."""246 self._exec_cmd("pkill -f 'node server.js'")247 248 # ==================================================================249 # TASK PROMPTS250 # ==================================================================251 def _build_task_prompt(self, init_stdout: str) -> str:252 """Build the task prompt based on the current difficulty level."""253 base = (254 "=== DEVOPS INCIDENT RESPONSE ===\n"255 f"ALERT: Production Node.js service in {self._app_dir} is DOWN.\n"256 "You are the on-call engineer. Diagnose and fix the issue(s).\n\n"257 "The app is an Express.js backend with multiple routes, middleware,\n"258 "config files, and logs. Not everything you see is broken — some files\n"259 "are red herrings. Focus on what's actually causing failures.\n\n"260 )261 262 if self._current_task == "easy":263 mission = (264 "SEVERITY: LOW (1 known issue)\n"265 "SYMPTOM: App fails to bind to the expected port.\n"266 "EXPECTED: App should listen on port 3000, GET /health returns 200.\n\n"267 "Start by checking configuration and trying to start the app.\n"268 )269 elif self._current_task == "medium":270 mission = (271 "SEVERITY: MEDIUM (2 known issues)\n"272 "SYMPTOMS:\n"273 " - App crashes immediately on startup\n"274 " - Even after fixing the crash, some routes may not work\n"275 "EXPECTED:\n"276 " - App listens on port 3000\n"277 " - GET /health returns 200\n"278 " - GET /api/users returns 200 with valid JSON\n\n"279 "Check startup logs carefully. The crash message will point you\n"280 "to the first bug, but there may be a config issue too.\n"281 )282 else:283 mission = (284 "SEVERITY: HIGH (3 known issues)\n"285 "SYMPTOMS:\n"286 " - App crashes on startup with an error\n"287 " - Multiple endpoints return errors or bad data\n"288 " - There are misleading old logs in logs/error.log\n"289 "EXPECTED:\n"290 " - App listens on port 3000\n"291 " - GET /health returns 200\n"292 " - GET /api/users returns 200 with JSON containing 'users' array\n"293 " - GET /api/data returns 200 with JSON containing 'records' array\n\n"294 "WARNING: The app has middleware, config files, .env, and old logs.\n"295 "Not everything is broken — isolate the actual root causes.\n"296 )297 298 return (299 base + mission +300 "\nUse bash commands to explore, edit files, and test.\n"301 "When you think you've fixed everything, run: cd /app && npm start\n\n"302 f"--- INITIAL STATE ---\n{init_stdout}\n"303 )304 305 def _bugs_for_task(self) -> int:306 """Return the number of bugs for the current task difficulty."""307 return {"easy": 1, "medium": 2, "hard": 3}.get(self._current_task, 3)308 309 # ==================================================================310 # CD HANDLER311 # ==================================================================312 def _handle_cd(self, command: str) -> TerminalObservation:313 """Handle cd commands manually since subprocess.run is transient."""314 target = command[3:].strip()315 if target == "" or target == "~":316 new_dir = self._app_dir317 elif target.startswith("/"):318 new_dir = os.path.normpath(target)319 else:320 new_dir = os.path.normpath(os.path.join(self._current_dir, target))321 322 if os.path.isdir(new_dir):323 self._current_dir = new_dir324 stdout, stderr = "", ""325 else:326 stdout, stderr = "", f"bash: cd: {target}: No such file or directory"327 328 score, feedback = self._grade()329 reward = max(0.01, score - self._last_score)330 self._last_score = score331 episode_done = (score >= 0.99) or (self._state.step_count >= MAX_STEPS)332 333 return TerminalObservation(334 stdout=stdout,335 stderr=stderr,336 current_dir=self._current_dir,337 task_id=self._current_task,338 grader_score=score,339 grader_feedback=feedback,340 done=episode_done,341 reward=reward,342 metadata=self._build_metadata(),343 )344 345 # ==================================================================346 # METADATA & FILE TRACKING347 # ==================================================================348 def _build_metadata(self) -> Dict[str, Any]:349 """Build rich metadata for the current observation."""350 return {351 "episode_id": self._state.episode_id,352 "step": self._state.step_count,353 "task": self._current_task,354 "max_steps": MAX_STEPS,355 "bugs_total": self._bugs_for_task(),356 "files_modified": list(self._files_modified),357 "commands_count": len(self._commands_history),358 }359 360 def _snapshot_file_hashes(self) -> None:361 """Take a hash snapshot of all bug-related files for change detection."""362 self._file_hashes = {}363 for relative_path in BUG_FILES:364 full_path = os.path.join(self._app_dir, relative_path)365 if os.path.isfile(full_path):366 try:367 with open(full_path, "rb") as f:368 self._file_hashes[relative_path] = hashlib.md5(f.read()).hexdigest()369 except OSError:370 pass371 372 def _detect_file_changes(self) -> None:373 """Detect which bug files have been modified since reset."""374 for relative_path in BUG_FILES:375 if relative_path in self._files_modified:376 continue377 full_path = os.path.join(self._app_dir, relative_path)378 if os.path.isfile(full_path):379 try:380 with open(full_path, "rb") as f:381 current_hash = hashlib.md5(f.read()).hexdigest()382 if current_hash != self._file_hashes.get(relative_path):383 self._files_modified.append(relative_path)384 except OSError:385 pass386 387 # ==================================================================388 # FILESYSTEM & EXECUTION HELPERS389 # ==================================================================390 def _reset_filesystem(self) -> None:391 """Replace the working /app with the pristine backup."""392 os.makedirs(self._app_dir, exist_ok=True)393 394 # Clean contents of /app395 for item in os.listdir(self._app_dir):396 item_path = os.path.join(self._app_dir, item)397 if os.path.isdir(item_path):398 shutil.rmtree(item_path, ignore_errors=True)399 else:400 try:401 os.remove(item_path)402 except OSError:403 pass404 405 # Copy from backup406 if os.path.exists(self._app_backup_dir):407 for item in os.listdir(self._app_backup_dir):408 s = os.path.join(self._app_backup_dir, item)409 d = os.path.join(self._app_dir, item)410 if os.path.isdir(s):411 shutil.copytree(s, d, dirs_exist_ok=True)412 else:413 shutil.copy2(s, d)414 else:415 logger.warning(416 f"Backup directory {self._app_backup_dir} not found. "417 "Ensure Dockerfile copied simulated_app here."418 )419 420 def _exec_cmd(self, cmd: str, timeout: float = 30.0) -> str:421 """Execute command natively; return combined output."""422 stdout, stderr = self._exec_cmd_split(cmd, timeout)423 return (stdout + "\n" + stderr).strip()424 425 def _exec_cmd_split(self, cmd: str, timeout: float = 30.0) -> Tuple[str, str]:426 """Execute command natively; return (stdout, stderr)."""427 kwargs = {428 "cwd": self._current_dir,429 "shell": True,430 "capture_output": True,431 "timeout": timeout,432 }433 if sys.platform != "win32":434 kwargs["executable"] = "/bin/bash"435 436 try:437 result = subprocess.run(cmd, **kwargs)438 return (439 result.stdout.decode(errors="replace"),440 result.stderr.decode(errors="replace"),441 )442 except subprocess.TimeoutExpired:443 return ("", "[command timed out]")444 except Exception as e:445 return ("", f"[exec error: {e}]")446 447 # ==================================================================448 # GRADER449 # ==================================================================450 def _inject_grader_script(self) -> None:451 """Write the grader bash script that tests the Node.js app endpoints."""452 self.grader_path = os.path.join(self._tmp_dir, "grader.sh")453 lines = [454 '#!/bin/bash',455 'set -m',456 '',457 'pkill -f "node server.js" 2>/dev/null',458 'sleep 0.5',459 '',460 f'cd {self._app_dir}',461 f'node server.js > {self._tmp_dir}/node.log 2>&1 &',462 'NODE_PID=$!',463 '',464 '# Wait for server to start (up to 4 seconds)',465 'for i in 1 2 3 4; do',466 ' sleep 1',467 ' if curl -s http://localhost:3000/health > /dev/null 2>&1; then',468 ' break',469 ' fi',470 'done',471 '',472 f'STARTUP_LOG=$(cat {self._tmp_dir}/node.log 2>/dev/null)',473 '',474 f"HEALTH_CODE=$(curl -s -o {self._tmp_dir}/health.json -w '%{{http_code}}' http://localhost:3000/health 2>/dev/null)",475 f"USERS_CODE=$(curl -s -o {self._tmp_dir}/users.json -w '%{{http_code}}' http://localhost:3000/api/users 2>/dev/null)",476 f"DATA_CODE=$(curl -s -o {self._tmp_dir}/data.json -w '%{{http_code}}' http://localhost:3000/api/data 2>/dev/null)",477 f'USERS_BODY=$(cat {self._tmp_dir}/users.json 2>/dev/null)',478 f'DATA_BODY=$(cat {self._tmp_dir}/data.json 2>/dev/null)',479 '',480 'kill $NODE_PID 2>/dev/null',481 'wait $NODE_PID 2>/dev/null',482 '',483 'echo "GRADER_STARTUP_LOG:${STARTUP_LOG}"',484 'echo "GRADER_HEALTH_CODE:${HEALTH_CODE}"',485 'echo "GRADER_USERS_CODE:${USERS_CODE}"',486 'echo "GRADER_DATA_CODE:${DATA_CODE}"',487 'echo "GRADER_USERS_BODY:${USERS_BODY}"',488 'echo "GRADER_DATA_BODY:${DATA_BODY}"',489 ]490 491 script_content = '\n'.join(lines) + '\n'492 with open(self.grader_path, "w", newline='\n') as f:493 f.write(script_content)494 495 if sys.platform != "win32":496 subprocess.run(["chmod", "+x", self.grader_path])497 498 def _grade(self) -> Tuple[float, str]:499 """Run the grader and return (score, feedback).500 501 Scoring breakdown:502 - File-level: +0.05 per correctly modified bug file503 - App starts on port 3000: +0.30504 - /health returns 200: +0.10505 - /api/users returns valid JSON: +0.15506 - /api/data returns valid JSON: +0.20507 - All endpoints pass: +0.05 bonus508 509 Total raw score is then scaled by task difficulty and clamped to (0, 1).510 """511 score = 0.0512 feedback_parts = []513 514 # --- Phase 1: File-change rewards (micro-rewards for finding bugs) ---515 files_to_check = {516 "easy": ["config.json"],517 "medium": ["config.json", "routes/users.js"],518 "hard": ["config.json", "routes/users.js", "routes/data.js"],519 }.get(self._current_task, list(BUG_FILES.keys()))520 521 for f in files_to_check:522 if f in self._files_modified:523 score += 0.05524 feedback_parts.append(f"✓ Modified {f} (+0.05)")525 526 # --- Phase 2: HTTP endpoint testing ---527 try:528 if sys.platform == "win32":529 raw = self._exec_cmd(f"bash {self.grader_path}", timeout=20.0)530 else:531 raw = self._exec_cmd(f"/bin/bash {self.grader_path}", timeout=20.0)532 533 results = {}534 for line in raw.splitlines():535 if line.startswith("GRADER_"):536 key, _, value = line.partition(":")537 results[key] = value.strip()538 539 startup_log = results.get("GRADER_STARTUP_LOG", "")540 health_code = results.get("GRADER_HEALTH_CODE", "000")541 users_code = results.get("GRADER_USERS_CODE", "000")542 data_code = results.get("GRADER_DATA_CODE", "000")543 users_body = results.get("GRADER_USERS_BODY", "")544 data_body = results.get("GRADER_DATA_BODY", "")545 546 has_syntax_error = "SyntaxError" in startup_log547 has_crash = (548 has_syntax_error549 or "Cannot find module" in startup_log550 or "ReferenceError" in startup_log551 )552 app_listening = f"Server running on port {EXPECTED_PORT}" in startup_log553 554 if has_crash and not app_listening:555 feedback_parts.append("✗ App crashes on startup")556 if has_syntax_error:557 feedback_parts.append("(SyntaxError detected)")558 # Fall through to clamping — NO early return559 elif not app_listening:560 feedback_parts.append("✗ App not listening on port 3000")561 # Fall through to clamping — NO early return562 else:563 # App is running — grade each endpoint564 score += 0.30565 feedback_parts.append("✓ App starts on port 3000 (+0.30)")566 567 if health_code == "200":568 score += 0.10569 feedback_parts.append("✓ /health returns 200 (+0.10)")570 else:571 feedback_parts.append(f"✗ /health returned {health_code}")572 573 if users_code == "200":574 if '"users"' in users_body:575 score += 0.15576 feedback_parts.append("✓ /api/users returns valid JSON (+0.15)")577 else:578 score += 0.05579 feedback_parts.append("~ /api/users 200 but malformed body (+0.05)")580 else:581 feedback_parts.append(f"✗ /api/users returned {users_code}")582 583 if data_code == "200":584 if '"records"' in data_body:585 score += 0.20586 feedback_parts.append("✓ /api/data returns valid JSON (+0.20)")587 else:588 score += 0.05589 feedback_parts.append("~ /api/data 200 but malformed body (+0.05)")590 else:591 feedback_parts.append(f"✗ /api/data returned {data_code}")592 593 if score >= 0.80:594 score += 0.05595 feedback_parts.append("✓ All endpoints healthy — bonus (+0.05)")596 597 except Exception as exc:598 logger.exception("Grader error")599 feedback_parts.append(f"Grader error (score preserved): {exc}")600 601 # --- Phase 3: Scale by difficulty and clamp ---602 if self._current_task == "easy":603 raw_target = 0.50604 elif self._current_task == "medium":605 raw_target = 0.65606 else:607 raw_target = 1.0608 609 final_score = min(1.0, score / raw_target)610 # Clamp strictly within (0, 1) — EVERY code path reaches here611 final_score = round(min(max(final_score, 0.01), 0.99), 2)612 613 return (final_score, " | ".join(feedback_parts))614 