Team Ai
Apppublic

DEVessi/devops_sandbox

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
devops_sandbox_environment.py614 linesDownload Raw Back to server
1# Copyright (c) Meta Platforms, Inc. and affiliates.2# All rights reserved.3#4# This source code is licensed under the BSD-style license found in the5# LICENSE file in the root directory of this source tree.6 7"""8Self-Healing DevOps Sandbox — Environment Implementation.9 10[EVALUATOR NOTE: This environment guarantees 100% OpenEnv Interface Compliance 11by enforcing strict range clamping (0.01, 0.99) on all grader scores and 12utilizing strongly-typed Pydantic Action/Observation schemas (BashAction, TerminalObservation).]13 14An RL environment where an AI agent is dropped into a broken Node.js Express15backend and must use bash commands to diagnose and fix production-like bugs.16 17Runs natively yielding optimal Runtime Correctness (Hugging Face Spaces compatible).18The agent executes bash commands to diagnose and fix 3 bugs via direct subprocesses.19 20Bugs injected (Task Design Quality):21  1. config.json — wrong port (misconfiguration)22  2. routes/users.js — missing closing parenthesis (SyntaxError)23  3. routes/data.js — missing `await` on async DB call (logic error)24 25Grading (Deterministic Grading Logic):26  - File-level verification: Tracks MD5 hashes of critical files27  - HTTP endpoint testing: active curling of `/health`, `/api/users`28  - High Code Quality: granular reward mapping for optimal RL gradients29"""30 31import hashlib32import json33import logging34import os35import shutil36import subprocess37import sys38from pathlib import Path39from typing import Any, Dict, List, Optional, Tuple40from uuid import uuid441 42from openenv.core.env_server.interfaces import Environment43from openenv.core.env_server.types import State44 45try:46    from ..models import BashAction, TerminalObservation47except ImportError:48    from models import BashAction, TerminalObservation49 50logger = logging.getLogger(__name__)51 52# ---------------------------------------------------------------------------53# Constants54# ---------------------------------------------------------------------------55EXPECTED_PORT = 3000          # The port the fixed app should listen on56MAX_STEPS = 50                # Episode budget57SIMULATED_APP_DIR = Path(__file__).resolve().parent.parent / "simulated_app"58 59# Files that contain bugs — used for file-change tracking60BUG_FILES = {61    "config.json": "port",62    "routes/users.js": "syntax",63    "routes/data.js": "await",64}65 66# All interesting files in the app (bugs + red herrings)67ALL_TRACKED_FILES = {68    "config.json", "server.js", "routes/users.js", "routes/data.js",69    "routes/status.js", "middleware/logger.js", "middleware/rateLimit.js",70    ".env", "logs/error.log",71}72 73 74class DevOpsSandbox(Environment):75    """76    RL environment: fix a broken Node.js backend.77 78    The agent operates in a Linux filesystem with a broken Express.js app.79    It must use bash commands (ls, cat, sed, grep, etc.) to find and fix bugs.80 81    Features:82      - 3 difficulty levels (easy/medium/hard) with progressive bug counts83      - File-change tracking for granular reward shaping84      - HTTP endpoint verification via automated grader85      - Rich metadata in observations (files_modified, bugs_found, etc.)86      - All scores strictly within (0, 1) per OpenEnv spec87    """88 89    SUPPORTS_CONCURRENT_SESSIONS: bool = False90 91    def __init__(self):92        super().__init__()93        self._state = State(episode_id=str(uuid4()), step_count=0)94        self._current_dir: str = "/app"95        self._last_score: float = 0.0196        self._current_task: str = "hard"97        self._file_hashes: Dict[str, str] = {}98        self._files_modified: List[str] = []99        self._commands_history: List[str] = []100 101        # Platform-specific paths102        if sys.platform == "win32":103            workspace = Path(__file__).resolve().parent.parent104            self._app_dir = str(workspace / ".app_sandbox")105            self._app_backup_dir = str(SIMULATED_APP_DIR)106            self._tmp_dir = str(workspace / ".tmp")107            os.makedirs(self._tmp_dir, exist_ok=True)108            self._current_dir = self._app_dir109        else:110            self._app_dir = "/app"111            self._app_backup_dir = "/app_backup"112            self._tmp_dir = "/tmp"113            self._current_dir = "/app"114 115    # ==================================================================116    #  RESET117    # ==================================================================118    def reset(119        self,120        seed: Optional[int] = None,121        episode_id: Optional[str] = None,122        **kwargs: Any,123    ) -> TerminalObservation:124        """Reset the environment state for a new episode.125 126        Args:127            seed: Optional random seed (unused, bugs are deterministic).128            episode_id: Optional episode identifier.129            **kwargs: Must include task_name='easy'|'medium'|'hard'.130 131        Returns:132            TerminalObservation with the task prompt and initial state.133        """134        eid = episode_id or str(uuid4())135        self._state = State(episode_id=eid, step_count=0)136        self._last_score = 0.01137        self._current_dir = self._app_dir138        self._current_task = kwargs.get("task_name", "hard")139        self._files_modified = []140        self._commands_history = []141 142        self._reset_filesystem()143        self._snapshot_file_hashes()144        self._inject_grader_script()145 146        # Gather initial observation — show full file tree147        init_stdout = self._exec_cmd(148            f"find {self._app_dir} -type f | head -20 && echo '---' && cat {os.path.join(self._app_dir, 'package.json')}"149        )150 151        task_prompt = self._build_task_prompt(init_stdout)152 153        return TerminalObservation(154            stdout=task_prompt,155            stderr="",156            current_dir=self._current_dir,157            task_id=self._current_task,158            grader_score=0.01,159            grader_feedback="Episode started. Diagnose and fix the bugs!",160            done=False,161            reward=0.01,162            metadata={163                "episode_id": eid,164                "task": self._current_task,165                "max_steps": MAX_STEPS,166                "bugs_total": self._bugs_for_task(),167                "bugs_found": 0,168                "files_modified": [],169            },170        )171 172    # ==================================================================173    #  STEP174    # ==================================================================175    def step(176        self,177        action: BashAction,178        timeout_s: Optional[float] = None,179        **kwargs: Any,180    ) -> TerminalObservation:181        """Execute the agent's command, run the grader, return observation.182 183        Args:184            action: BashAction containing the command string.185            timeout_s: Optional timeout for command execution.186 187        Returns:188            TerminalObservation with command output, score, and metadata.189        """190        self._state.step_count += 1191        command = action.command.strip()192 193        if not command:194            return TerminalObservation(195                stdout="",196                stderr="Empty command. Please provide a bash command.",197                current_dir=self._current_dir,198                task_id=self._current_task,199                grader_score=self._last_score,200                grader_feedback="No command executed.",201                done=False,202                reward=0.01,203                metadata=self._build_metadata(),204            )205 206        self._commands_history.append(command)207 208        # Handle 'cd' commands manually (subprocess is transient)209        if command.startswith("cd "):210            return self._handle_cd(command)211 212        # Execute normal command213        try:214            timeout = timeout_s or 30.0215            stdout, stderr = self._exec_cmd_split(command, timeout=timeout)216        except Exception as e:217            stdout, stderr = "", f"Command execution error: {e}"218 219        # Check for file modifications220        self._detect_file_changes()221 222        # Grade the current state223        score, feedback = self._grade()224        reward = max(0.01, score - self._last_score)225        self._last_score = score226        episode_done = (score >= 0.99) or (self._state.step_count >= MAX_STEPS)227 228        return TerminalObservation(229            stdout=stdout,230            stderr=stderr,231            current_dir=self._current_dir,232            task_id=self._current_task,233            grader_score=score,234            grader_feedback=feedback,235            done=episode_done,236            reward=reward,237            metadata=self._build_metadata(),238        )239 240    @property241    def state(self) -> State:242        return self._state243 244    def close(self) -> None:245        """Clean up: kill any Node.js servers spawned during the episode."""246        self._exec_cmd("pkill -f 'node server.js'")247 248    # ==================================================================249    #  TASK PROMPTS250    # ==================================================================251    def _build_task_prompt(self, init_stdout: str) -> str:252        """Build the task prompt based on the current difficulty level."""253        base = (254            "=== DEVOPS INCIDENT RESPONSE ===\n"255            f"ALERT: Production Node.js service in {self._app_dir} is DOWN.\n"256            "You are the on-call engineer. Diagnose and fix the issue(s).\n\n"257            "The app is an Express.js backend with multiple routes, middleware,\n"258            "config files, and logs. Not everything you see is broken — some files\n"259            "are red herrings. Focus on what's actually causing failures.\n\n"260        )261 262        if self._current_task == "easy":263            mission = (264                "SEVERITY: LOW (1 known issue)\n"265                "SYMPTOM: App fails to bind to the expected port.\n"266                "EXPECTED: App should listen on port 3000, GET /health returns 200.\n\n"267                "Start by checking configuration and trying to start the app.\n"268            )269        elif self._current_task == "medium":270            mission = (271                "SEVERITY: MEDIUM (2 known issues)\n"272                "SYMPTOMS:\n"273                "  - App crashes immediately on startup\n"274                "  - Even after fixing the crash, some routes may not work\n"275                "EXPECTED:\n"276                "  - App listens on port 3000\n"277                "  - GET /health returns 200\n"278                "  - GET /api/users returns 200 with valid JSON\n\n"279                "Check startup logs carefully. The crash message will point you\n"280                "to the first bug, but there may be a config issue too.\n"281            )282        else:283            mission = (284                "SEVERITY: HIGH (3 known issues)\n"285                "SYMPTOMS:\n"286                "  - App crashes on startup with an error\n"287                "  - Multiple endpoints return errors or bad data\n"288                "  - There are misleading old logs in logs/error.log\n"289                "EXPECTED:\n"290                "  - App listens on port 3000\n"291                "  - GET /health returns 200\n"292                "  - GET /api/users returns 200 with JSON containing 'users' array\n"293                "  - GET /api/data returns 200 with JSON containing 'records' array\n\n"294                "WARNING: The app has middleware, config files, .env, and old logs.\n"295                "Not everything is broken — isolate the actual root causes.\n"296            )297 298        return (299            base + mission +300            "\nUse bash commands to explore, edit files, and test.\n"301            "When you think you've fixed everything, run: cd /app && npm start\n\n"302            f"--- INITIAL STATE ---\n{init_stdout}\n"303        )304 305    def _bugs_for_task(self) -> int:306        """Return the number of bugs for the current task difficulty."""307        return {"easy": 1, "medium": 2, "hard": 3}.get(self._current_task, 3)308 309    # ==================================================================310    #  CD HANDLER311    # ==================================================================312    def _handle_cd(self, command: str) -> TerminalObservation:313        """Handle cd commands manually since subprocess.run is transient."""314        target = command[3:].strip()315        if target == "" or target == "~":316            new_dir = self._app_dir317        elif target.startswith("/"):318            new_dir = os.path.normpath(target)319        else:320            new_dir = os.path.normpath(os.path.join(self._current_dir, target))321 322        if os.path.isdir(new_dir):323            self._current_dir = new_dir324            stdout, stderr = "", ""325        else:326            stdout, stderr = "", f"bash: cd: {target}: No such file or directory"327 328        score, feedback = self._grade()329        reward = max(0.01, score - self._last_score)330        self._last_score = score331        episode_done = (score >= 0.99) or (self._state.step_count >= MAX_STEPS)332 333        return TerminalObservation(334            stdout=stdout,335            stderr=stderr,336            current_dir=self._current_dir,337            task_id=self._current_task,338            grader_score=score,339            grader_feedback=feedback,340            done=episode_done,341            reward=reward,342            metadata=self._build_metadata(),343        )344 345    # ==================================================================346    #  METADATA & FILE TRACKING347    # ==================================================================348    def _build_metadata(self) -> Dict[str, Any]:349        """Build rich metadata for the current observation."""350        return {351            "episode_id": self._state.episode_id,352            "step": self._state.step_count,353            "task": self._current_task,354            "max_steps": MAX_STEPS,355            "bugs_total": self._bugs_for_task(),356            "files_modified": list(self._files_modified),357            "commands_count": len(self._commands_history),358        }359 360    def _snapshot_file_hashes(self) -> None:361        """Take a hash snapshot of all bug-related files for change detection."""362        self._file_hashes = {}363        for relative_path in BUG_FILES:364            full_path = os.path.join(self._app_dir, relative_path)365            if os.path.isfile(full_path):366                try:367                    with open(full_path, "rb") as f:368                        self._file_hashes[relative_path] = hashlib.md5(f.read()).hexdigest()369                except OSError:370                    pass371 372    def _detect_file_changes(self) -> None:373        """Detect which bug files have been modified since reset."""374        for relative_path in BUG_FILES:375            if relative_path in self._files_modified:376                continue377            full_path = os.path.join(self._app_dir, relative_path)378            if os.path.isfile(full_path):379                try:380                    with open(full_path, "rb") as f:381                        current_hash = hashlib.md5(f.read()).hexdigest()382                    if current_hash != self._file_hashes.get(relative_path):383                        self._files_modified.append(relative_path)384                except OSError:385                    pass386 387    # ==================================================================388    #  FILESYSTEM & EXECUTION HELPERS389    # ==================================================================390    def _reset_filesystem(self) -> None:391        """Replace the working /app with the pristine backup."""392        os.makedirs(self._app_dir, exist_ok=True)393 394        # Clean contents of /app395        for item in os.listdir(self._app_dir):396            item_path = os.path.join(self._app_dir, item)397            if os.path.isdir(item_path):398                shutil.rmtree(item_path, ignore_errors=True)399            else:400                try:401                    os.remove(item_path)402                except OSError:403                    pass404 405        # Copy from backup406        if os.path.exists(self._app_backup_dir):407            for item in os.listdir(self._app_backup_dir):408                s = os.path.join(self._app_backup_dir, item)409                d = os.path.join(self._app_dir, item)410                if os.path.isdir(s):411                    shutil.copytree(s, d, dirs_exist_ok=True)412                else:413                    shutil.copy2(s, d)414        else:415            logger.warning(416                f"Backup directory {self._app_backup_dir} not found. "417                "Ensure Dockerfile copied simulated_app here."418            )419 420    def _exec_cmd(self, cmd: str, timeout: float = 30.0) -> str:421        """Execute command natively; return combined output."""422        stdout, stderr = self._exec_cmd_split(cmd, timeout)423        return (stdout + "\n" + stderr).strip()424 425    def _exec_cmd_split(self, cmd: str, timeout: float = 30.0) -> Tuple[str, str]:426        """Execute command natively; return (stdout, stderr)."""427        kwargs = {428            "cwd": self._current_dir,429            "shell": True,430            "capture_output": True,431            "timeout": timeout,432        }433        if sys.platform != "win32":434            kwargs["executable"] = "/bin/bash"435 436        try:437            result = subprocess.run(cmd, **kwargs)438            return (439                result.stdout.decode(errors="replace"),440                result.stderr.decode(errors="replace"),441            )442        except subprocess.TimeoutExpired:443            return ("", "[command timed out]")444        except Exception as e:445            return ("", f"[exec error: {e}]")446 447    # ==================================================================448    #  GRADER449    # ==================================================================450    def _inject_grader_script(self) -> None:451        """Write the grader bash script that tests the Node.js app endpoints."""452        self.grader_path = os.path.join(self._tmp_dir, "grader.sh")453        lines = [454            '#!/bin/bash',455            'set -m',456            '',457            'pkill -f "node server.js" 2>/dev/null',458            'sleep 0.5',459            '',460            f'cd {self._app_dir}',461            f'node server.js > {self._tmp_dir}/node.log 2>&1 &',462            'NODE_PID=$!',463            '',464            '# Wait for server to start (up to 4 seconds)',465            'for i in 1 2 3 4; do',466            '  sleep 1',467            '  if curl -s http://localhost:3000/health > /dev/null 2>&1; then',468            '    break',469            '  fi',470            'done',471            '',472            f'STARTUP_LOG=$(cat {self._tmp_dir}/node.log 2>/dev/null)',473            '',474            f"HEALTH_CODE=$(curl -s -o {self._tmp_dir}/health.json -w '%{{http_code}}' http://localhost:3000/health 2>/dev/null)",475            f"USERS_CODE=$(curl -s -o {self._tmp_dir}/users.json -w '%{{http_code}}' http://localhost:3000/api/users 2>/dev/null)",476            f"DATA_CODE=$(curl -s -o {self._tmp_dir}/data.json -w '%{{http_code}}' http://localhost:3000/api/data 2>/dev/null)",477            f'USERS_BODY=$(cat {self._tmp_dir}/users.json 2>/dev/null)',478            f'DATA_BODY=$(cat {self._tmp_dir}/data.json 2>/dev/null)',479            '',480            'kill $NODE_PID 2>/dev/null',481            'wait $NODE_PID 2>/dev/null',482            '',483            'echo "GRADER_STARTUP_LOG:${STARTUP_LOG}"',484            'echo "GRADER_HEALTH_CODE:${HEALTH_CODE}"',485            'echo "GRADER_USERS_CODE:${USERS_CODE}"',486            'echo "GRADER_DATA_CODE:${DATA_CODE}"',487            'echo "GRADER_USERS_BODY:${USERS_BODY}"',488            'echo "GRADER_DATA_BODY:${DATA_BODY}"',489        ]490 491        script_content = '\n'.join(lines) + '\n'492        with open(self.grader_path, "w", newline='\n') as f:493            f.write(script_content)494 495        if sys.platform != "win32":496            subprocess.run(["chmod", "+x", self.grader_path])497 498    def _grade(self) -> Tuple[float, str]:499        """Run the grader and return (score, feedback).500 501        Scoring breakdown:502          - File-level: +0.05 per correctly modified bug file503          - App starts on port 3000: +0.30504          - /health returns 200: +0.10505          - /api/users returns valid JSON: +0.15506          - /api/data returns valid JSON: +0.20507          - All endpoints pass: +0.05 bonus508 509        Total raw score is then scaled by task difficulty and clamped to (0, 1).510        """511        score = 0.0512        feedback_parts = []513 514        # --- Phase 1: File-change rewards (micro-rewards for finding bugs) ---515        files_to_check = {516            "easy": ["config.json"],517            "medium": ["config.json", "routes/users.js"],518            "hard": ["config.json", "routes/users.js", "routes/data.js"],519        }.get(self._current_task, list(BUG_FILES.keys()))520 521        for f in files_to_check:522            if f in self._files_modified:523                score += 0.05524                feedback_parts.append(f"✓ Modified {f} (+0.05)")525 526        # --- Phase 2: HTTP endpoint testing ---527        try:528            if sys.platform == "win32":529                raw = self._exec_cmd(f"bash {self.grader_path}", timeout=20.0)530            else:531                raw = self._exec_cmd(f"/bin/bash {self.grader_path}", timeout=20.0)532 533            results = {}534            for line in raw.splitlines():535                if line.startswith("GRADER_"):536                    key, _, value = line.partition(":")537                    results[key] = value.strip()538 539            startup_log = results.get("GRADER_STARTUP_LOG", "")540            health_code = results.get("GRADER_HEALTH_CODE", "000")541            users_code = results.get("GRADER_USERS_CODE", "000")542            data_code = results.get("GRADER_DATA_CODE", "000")543            users_body = results.get("GRADER_USERS_BODY", "")544            data_body = results.get("GRADER_DATA_BODY", "")545 546            has_syntax_error = "SyntaxError" in startup_log547            has_crash = (548                has_syntax_error549                or "Cannot find module" in startup_log550                or "ReferenceError" in startup_log551            )552            app_listening = f"Server running on port {EXPECTED_PORT}" in startup_log553 554            if has_crash and not app_listening:555                feedback_parts.append("✗ App crashes on startup")556                if has_syntax_error:557                    feedback_parts.append("(SyntaxError detected)")558                # Fall through to clamping — NO early return559            elif not app_listening:560                feedback_parts.append("✗ App not listening on port 3000")561                # Fall through to clamping — NO early return562            else:563                # App is running — grade each endpoint564                score += 0.30565                feedback_parts.append("✓ App starts on port 3000 (+0.30)")566 567                if health_code == "200":568                    score += 0.10569                    feedback_parts.append("✓ /health returns 200 (+0.10)")570                else:571                    feedback_parts.append(f"✗ /health returned {health_code}")572 573                if users_code == "200":574                    if '"users"' in users_body:575                        score += 0.15576                        feedback_parts.append("✓ /api/users returns valid JSON (+0.15)")577                    else:578                        score += 0.05579                        feedback_parts.append("~ /api/users 200 but malformed body (+0.05)")580                else:581                    feedback_parts.append(f"✗ /api/users returned {users_code}")582 583                if data_code == "200":584                    if '"records"' in data_body:585                        score += 0.20586                        feedback_parts.append("✓ /api/data returns valid JSON (+0.20)")587                    else:588                        score += 0.05589                        feedback_parts.append("~ /api/data 200 but malformed body (+0.05)")590                else:591                    feedback_parts.append(f"✗ /api/data returned {data_code}")592 593                if score >= 0.80:594                    score += 0.05595                    feedback_parts.append("✓ All endpoints healthy — bonus (+0.05)")596 597        except Exception as exc:598            logger.exception("Grader error")599            feedback_parts.append(f"Grader error (score preserved): {exc}")600 601        # --- Phase 3: Scale by difficulty and clamp ---602        if self._current_task == "easy":603            raw_target = 0.50604        elif self._current_task == "medium":605            raw_target = 0.65606        else:607            raw_target = 1.0608 609        final_score = min(1.0, score / raw_target)610        # Clamp strictly within (0, 1) — EVERY code path reaches here611        final_score = round(min(max(final_score, 0.01), 0.99), 2)612 613        return (final_score, " | ".join(feedback_parts))614