DEVessi/devops_sandbox
0
1#!/usr/bin/env python32# Copyright (c) Meta Platforms, Inc. and affiliates.3# All rights reserved.4#5# This source code is licensed under the BSD-style license found in the6# LICENSE file in the root directory of this source tree.7 8"""9Baseline inference script for the Self-Healing DevOps Sandbox.10 11Uses an LLM (via the OpenAI-compatible API) to diagnose and fix a broken12Node.js backend running inside a Docker container.13"""14 15import json16import os17import sys18 19try:20 from openai import OpenAI21except ImportError:22 print("ERROR: 'openai' package is required. Install with: pip install openai")23 sys.exit(1)24 25from client import DevopsSandboxEnv26from models import BashAction27 28# ---------------------------------------------------------------------------29# Configuration30# ---------------------------------------------------------------------------31API_BASE_URL = os.getenv("API_BASE_URL") or "https://router.huggingface.co/v1"32MODEL_NAME = os.getenv("MODEL_NAME") or "gpt-4o-mini"33HF_TOKEN = os.getenv("HF_TOKEN") or os.getenv("API_KEY")34 35ENV_URL = os.getenv("DEVOPS_SANDBOX_URL", "http://localhost:8000")36TASK_NAME = os.getenv("MY_ENV_V4_TASK", "devops_sandbox")37BENCHMARK = os.getenv("MY_ENV_V4_BENCHMARK", "devops_sandbox")38MAX_TURNS = int(os.getenv("MAX_TURNS", "8"))39 40SYSTEM_PROMPT = """\41You are an expert DevOps engineer and Node.js developer.42 43You have been dropped into a Linux container with a broken Express.js backend in /app.44Your goal is to diagnose and fix ALL bugs so the app runs correctly.45 46RULES:471. Respond ONLY with a JSON object: {"command": "<bash command>"}482. Use standard bash/Linux commands (ls, cat, grep, sed, node, npm, etc.)493. Do NOT use interactive editors (vi, nano). Use sed or echo/cat with redirection.504. After fixing bugs, restart the app with: cd /app && npm start &515. Be methodical: read files first, understand the bug, then fix it.52 53EXPECTED FINAL STATE:54- App starts without errors on port 300055- GET /health → 20056- GET /api/users → 200 with JSON containing "users" array57- GET /api/data → 200 with JSON containing "records" array58"""59 60def extract_command(llm_response: str) -> str:61 """Extract a bash command from the LLM's response (JSON or raw text)."""62 try:63 data = json.loads(llm_response.strip())64 if isinstance(data, dict) and "command" in data:65 return data["command"]66 except (json.JSONDecodeError, TypeError):67 pass68 69 if "```" in llm_response:70 lines = llm_response.split("```")71 for block in lines[1::2]:72 code = block.strip()73 if code.startswith("json"):74 code = code[4:].strip()75 try:76 data = json.loads(code)77 if isinstance(data, dict) and "command" in data:78 return data["command"]79 except (json.JSONDecodeError, TypeError):80 pass81 elif code.startswith("bash") or code.startswith("sh"):82 code = code.split("\n", 1)[-1].strip()83 return code84 else:85 first_line = code.split("\n")[0].strip()86 if first_line:87 return first_line88 89 cmd = llm_response.strip().strip("`").strip()90 if cmd.startswith("{"):91 try:92 return json.loads(cmd)["command"]93 except Exception:94 pass95 return cmd96 97def main():98 if not HF_TOKEN:99 pass # we can let it fail or use empty key depending on endpoint100 101 client = OpenAI(api_key=HF_TOKEN or "dummy_key", base_url=API_BASE_URL)102 103 TASKS = ["easy", "medium", "hard"]104 105 # Note: openenv evaluation specifically needs exactly 3 things: [START], [STEP] logs, [END]106 for task_name in TASKS:107 messages = [{"role": "system", "content": SYSTEM_PROMPT}]108 109 try:110 with DevopsSandboxEnv(base_url=ENV_URL).sync() as env:111 result = env.reset(task_name=task_name)112 obs = result.observation113 114 print(f"[START] task={task_name} env={BENCHMARK} model={MODEL_NAME}", flush=True)115 116 messages.append({117 "role": "user",118 "content": (119 f"Here is the initial state of the broken app:\n\n"120 f"```\n{obs.stdout}\n```\n\n"121 f"Current directory: {obs.current_dir}\n"122 f"Score: {obs.grader_score}/1.0\n\n"123 f"What bash command should I run first?"124 ),125 })126 127 rewards = []128 is_done = False129 steps_taken = 0130 final_score = getattr(obs, 'grader_score', 0.01)131 132 for turn in range(1, MAX_TURNS + 1):133 try:134 response = client.chat.completions.create(135 model=MODEL_NAME,136 messages=messages,137 temperature=0.2,138 max_tokens=256,139 )140 llm_text = response.choices[0].message.content or ""141 except Exception as e:142 err_msg = str(e).replace('"', "'")143 break144 145 command = extract_command(llm_text)146 if not command:147 command = "ls -la /app"148 149 error_msg = "null"150 try:151 result = env.step(BashAction(command=command))152 obs = result.observation153 except Exception as e:154 obs = env.state # Mock failed obs155 error_msg = str(e).replace('\n', ' ')156 157 steps_taken += 1158 reward_val = obs.reward if hasattr(obs, 'reward') else getattr(obs, 'grader_score', 0.01)159 rewards.append(f"{reward_val:.2f}")160 is_done = result.done if hasattr(result, 'done') else getattr(obs, 'done', False)161 done_str = "true" if is_done else "false"162 163 action_str = command.replace('\n', ' ; ')164 print(f"[STEP] step={steps_taken} action={action_str} reward={reward_val:.2f} done={done_str} error={error_msg}", flush=True)165 166 messages.append({"role": "assistant", "content": llm_text})167 messages.append({168 "role": "user",169 "content": (170 f"Command output:\n"171 f"stdout:\n```\n{getattr(obs, 'stdout', '')}\n```\n"172 f"stderr:\n```\n{getattr(obs, 'stderr', '')}\n```\n"173 f"Current score: {getattr(obs, 'grader_score', 0.01)}/1.0\n"174 f"Grader feedback: {getattr(obs, 'grader_feedback', '')}\n\n"175 f"What command should I run next?"176 ),177 })178 179 final_score = getattr(obs, 'grader_score', 0.01)180 if final_score >= 0.99 or getattr(obs, 'done', False) or (hasattr(result, 'done') and result.done):181 break182 183 # Clamp final score strictly within (0, 1)184 final_score = max(0.01, min(0.99, final_score))185 success_str = "true" if final_score >= 0.99 else "false"186 rewards_str = ",".join(rewards) if rewards else "0.01"187 print(f"[END] success={success_str} steps={steps_taken} score={final_score:.2f} rewards={rewards_str}", flush=True)188 except Exception as e:189 # Make sure to emit END log even on catastrophic wrapper failures so Hackathon doesn't crash inference.py190 print(f"[END] success=false steps=0 score=0.01 rewards=0.01", flush=True)191 192if __name__ == "__main__":193 main()194 