MrMoz33/tokioai-coder-iot
0
1"""2TokioAI Engine v3.0 -- OpenAI-Compatible API Server3SERVER-SIDE MULTI-STEP INVESTIGATION (the Opus architecture)4 5The key breakthrough: when the model calls a read-only tool (execute_local,6read_file, search_files, diagnose), the ENGINE executes it directly and feeds7the result back to the model -- all within a SINGLE API request. This means:8 91. User asks "is Home Assistant running?"102. Model calls execute_local(systemctl status home-assistant)113. Engine EXECUTES it -> "not found"124. Engine feeds result back to model135. Model tries execute_local(docker ps | grep -i home)146. Engine EXECUTES it -> "not found"157. Model tries execute_local(ss -tlnp | grep 8123)168. Engine EXECUTES it -> port 8123 listening179. Model tries execute_local(curl -s http://localhost:8123/api/)1810. Engine EXECUTES it -> HA API responds1911. Model answers: "Yes, HA is running on port 8123 via Docker"20 21All 10 steps happen in ONE API call. The CLI sees only the final answer.22For WRITE tools (write_file, edit_file), the Engine delegates to the CLI.23 24Endpoints:25 POST /v1/chat/completions -- Standard chat completion (with tool support)26 GET /v1/models -- List available models27 GET /health -- Health check28 GET /stats -- Engine statistics29"""30 31import json32import os33import sys34import time35import uuid36import argparse37import re38from typing import Dict, List, Optional, Tuple39 40from .pipeline import Pipeline, EngineConfig, PipelineResult41from .router import classify_intent42from .executor import execute_tool, is_safe_tool, SAFE_TOOLS, DELEGATE_TOOLS43 44 45def has_fake_tool_execution(text: str) -> bool:46 """Detect if text contains hallucinated/fake tool execution.47 48 ANY bash/shell code block with a system command means the model is 49 writing commands in text instead of using real tool_calls. This is 50 ALWAYS wrong -- the model should call execute_local, not print commands.51 """52 if not text:53 return False54 t = text.lower()55 # Pattern: ANY code block with a system command = fake tool execution56 # The model should be using tool_calls, not writing bash blocks57 if re.search(r'```(?:bash|shell|sh)?\s*\n\s*(?:ss|netstat|nmap|curl|ps|docker|find|grep|ls|cat|df|du|systemctl|journalctl|free|top|uptime|uname|id|ifconfig|ip|iptables|who|w|last|lastlog|users|finger|hostname|whoami|date|mount|lsblk|lsof|apt|dpkg|pip|npm|service|timedatectl|hostnamectl|sleep|HA_TOKEN)\b', t):58 return True59 if 'ejecutemos' in t or 'let me run' in t or 'voy a ejecutar' in t or 'voy a verificar' in t:60 if '```' in t:61 return True62 # Pattern: any code block immediately followed by "Result:" with fake data63 if re.search(r'```[^`]*```\s*\n*result\s*:', t):64 return True65 return False66 67 68def clean_model_output(text: str) -> str:69 """Remove model artifacts and hallucinated tool execution from output text."""70 if not text:71 return text72 # Remove trailing <tool_call> or [TOOL_CALLS] artifacts73 text = re.sub(r'</?tool_call>\s*$', '', text)74 text = re.sub(r'\[TOOL_CALLS?\]\s*$', '', text)75 text = re.sub(r'\[/TOOL_CALLS?\]\s*$', '', text)76 # Remove partial JSON tool call attempts at the end77 text = re.sub(r'\{"name"\s*:\s*"[^"]*"\s*,?\s*("arguments".*)?$', '', text)78 # Remove markdown-wrapped empty tool calls79 text = re.sub(r'```json\s*\{[^}]*\}\s*```\s*$', '', text)80 81 # Remove ALL code blocks containing system commands + fake output82 # This catches: ```bash\nss -tuln\n```\n\nResult: ...\n```...```83 text = re.sub(84 r'```(?:bash|shell|sh)?\s*\n\s*(?:ss|netstat|nmap|curl|ps|docker|find|grep|ls|cat|df|du|systemctl|journalctl|apt|pip|npm|which|whereis|locate|dpkg|service|mount|lsblk|lsof|free|top|htop|uptime|hostname|uname|whoami|id|ifconfig|ip|ping|traceroute|dig|nslookup|iptables|sudo|who|w|last|lastlog|users|date|timedatectl)\b[^`]*```',85 '', text, flags=re.MULTILINE | re.DOTALL86 )87 88 # Remove fake "Result:" sections (with or without code blocks)89 text = re.sub(90 r'(?:Result|Output|Salida)\s*:\s*(?:Active|Proto|tcp|udp|LISTEN|Filesystem|total|Mem:|Swap:).*?(?=\n\n[A-Z]|\Z)',91 '', text, flags=re.DOTALL | re.IGNORECASE92 )93 94 # Remove hallucinated inline tool execution blocks95 # Pattern: ```bash\nCOMMAND\n```\n\nResult of tool:\nFAKE_OUTPUT96 text = re.sub(97 r'```(?:bash|shell|sh)\n[^`]+```\s*\n*(?:Result of tool:?\s*\n[^\n]*(?:\n[^\n]*)*?)(?=\n\n|\Z)',98 '', text, flags=re.MULTILINE99 )100 # Pattern: bare ```\nCOMMAND\n``` blocks (no language tag)101 text = re.sub(102 r'```\n(?:(?:ss|netstat|nmap|curl|ps|docker|find|grep|ls|cat|df|du|systemctl|journalctl|apt|pip|npm|which|whereis|locate|dpkg|service|mount|lsblk|lsof|free|top|htop|uptime|hostname|uname|whoami|id|ifconfig|ip|ping|traceroute|dig|nslookup)\b[^`]*)\n```',103 '', text, flags=re.MULTILINE104 )105 # Pattern: "I ran: tool_name(...)\n[OK] Result:\n..."106 text = re.sub(107 r'I ran:\s*\w+\([^)]*\)\s*\n\[(?:OK|FAIL(?:ED)?)\]\s*Result:\n.*?(?=\n\n|\Z)',108 '', text, flags=re.DOTALL109 )110 # Pattern: standalone "Result of tool:" blocks111 text = re.sub(112 r'Result of tool:\s*\n.*?(?=\n\n|\Z)',113 '', text, flags=re.DOTALL114 )115 # Remove trailing lazy/filler phrases116 lazy_tails = [117 r'\n+(?:Let me know|If you\'?d? like|Do you want|Quieres que|Te gustaría|Si necesitas|Para más detalles).*$',118 r'\n+(?:Would you like me to|Want me to|If you want me to|Si quieres que).*$',119 ]120 for pattern in lazy_tails:121 text = re.sub(pattern, '', text, flags=re.IGNORECASE | re.DOTALL)122 123 # Clean up excessive blank lines left behind124 text = re.sub(r'\n{3,}', '\n\n', text)125 return text.strip()126 127 128# Max number of server-side investigation rounds per request129MAX_INVESTIGATION_DEPTH = 8130 131# Max total time for server-side investigation (seconds)132MAX_INVESTIGATION_TIME = 90133 134 135def count_tool_depth(messages: List[Dict]) -> int:136 """Count tool call rounds in messages."""137 depth = 0138 for msg in messages:139 if msg.get("role") == "assistant" and msg.get("tool_calls"):140 depth += 1141 return depth142 143 144def find_original_query(messages: List[Dict]) -> str:145 """Find the first user message (the original question)."""146 for msg in messages:147 if msg.get("role") == "user":148 content = msg.get("content", "")149 if content and not content.startswith("INVESTIGATION"):150 return content151 return ""152 153 154def create_app(config: EngineConfig = None):155 """Create the Flask app with the engine pipeline."""156 157 try:158 from flask import Flask, request, jsonify, Response159 except ImportError:160 print("Flask not installed. Install with: pip install flask")161 sys.exit(1)162 163 app = Flask("tokioai-engine")164 engine_config = config or EngineConfig()165 engine = Pipeline(engine_config)166 167 # Track investigation stats168 investigation_stats = {169 "total_investigations": 0,170 "total_steps": 0,171 "max_depth_reached": 0,172 "delegated_to_cli": 0,173 "avg_steps": 0,174 }175 176 @app.route("/health", methods=["GET"])177 def health():178 return jsonify({179 "status": "ok",180 "engine": "tokioai-engine",181 "version": "3.0.0",182 "model": engine.config.model_name,183 "ollama_host": engine.config.ollama_host,184 "max_investigation_depth": MAX_INVESTIGATION_DEPTH,185 "max_investigation_time": MAX_INVESTIGATION_TIME,186 "server_side_execution": True,187 })188 189 @app.route("/stats", methods=["GET"])190 def stats():191 s = engine.get_stats()192 s["investigation"] = investigation_stats193 return jsonify(s)194 195 @app.route("/v1/models", methods=["GET"])196 def list_models():197 return jsonify({198 "object": "list",199 "data": [200 {201 "id": "tokioai-engine",202 "object": "model",203 "created": int(time.time()),204 "owned_by": "tokioai",205 "permission": [],206 "root": engine.config.model_name,207 "parent": None,208 }209 ]210 })211 212 @app.route("/v1/chat/completions", methods=["POST"])213 def chat_completions():214 data = request.get_json()215 216 if not data:217 return jsonify({"error": {"message": "Request body is required"}}), 400218 219 messages = data.get("messages", [])220 if not messages:221 return jsonify({"error": {"message": "messages is required"}}), 400222 223 stream = data.get("stream", False)224 225 # ============================================================226 # v3.0 SERVER-SIDE INVESTIGATION LOOP227 # ============================================================228 #229 # Architecture:230 # 1. Extract user message and conversation history231 # 2. Call model (via Pipeline) -- model returns tool_calls or text232 # 3. If tool_calls:233 # a. If ALL tools are SAFE (read-only): execute server-side,234 # feed results back to model, goto step 2235 # b. If ANY tool is DELEGATE (write/remote): return tool_calls236 # to CLI for execution (old behavior)237 # 4. If text: return the text answer238 # 5. Repeat up to MAX_INVESTIGATION_DEPTH or MAX_INVESTIGATION_TIME239 240 # Determine if this is a tool-result continuation from CLI241 last_msg_role = None242 for m in reversed(messages):243 if m.get("role") != "system":244 last_msg_role = m.get("role")245 break246 247 is_tool_continuation = (last_msg_role == "tool")248 249 # Build the working message history250 user_msg, working_messages = _extract_messages(messages)251 252 if not user_msg:253 return jsonify({"error": {"message": "No user message found"}}), 400254 255 original_query = user_msg256 257 # If this is a CLI tool continuation (CLI executed a write/remote tool),258 # let the model see the result and decide next step259 if is_tool_continuation:260 # The CLI sent back results from a delegated tool.261 # Feed it all through and let the model continue.262 pass263 264 # === PRE-CLASSIFICATION: Skip investigation for pure text ===265 # For greetings, thanks -- skip the investigation loop entirely.266 # BUT NOT for affirmations ("dale", "si", "ok") which are contextual267 # and need conversation history to understand what the user is agreeing to.268 _GREETING_PATTERNS = [269 r'^(hi|hello|hey|hola|que\s+tal|buenos?\s+d[ií]as?|buenas?\s+(?:tardes?|noches?))\s*[.!?]*$',270 r'^(thanks?|thank\s+you|gracias|thx|de\s+nada)\s*[.!?]*$',271 r'^(como\s+(va|estas?|andas?|te\s+va)|how\s+are\s+you|how\'?s\s+it\s+going|what\'?s\s+up)\s*[.!?]*$',272 r'^(todo\s+bien|bien\s+y\s+tu|genial|cool|nice|great|awesome)\s*[.!?]*$',273 r'^(que\s+onda|que\s+hay|que\s+pasa|que\s+cuentas)\s*[.!?]*$',274 r'\b(who\s+are\s+you|what\s+can\s+you|tell\s+me\s+about\s+yourself)\b',275 ]276 import re as _re277 _is_pure_greeting = any(_re.match(p, user_msg.lower().strip(), _re.IGNORECASE) for p in _GREETING_PATTERNS)278 if not is_tool_continuation and _is_pure_greeting:279 _shortcut_start = time.time()280 if engine_config.verbose:281 print(f"[ENGINE] Pure text shortcut: '{user_msg}' -> greeting/thanks")282 engine.reset()283 result = engine.process(284 user_message=user_msg,285 conversation_history=working_messages,286 force_intent="text",287 )288 elapsed = time.time() - _shortcut_start289 return jsonify({290 "id": f"chatcmpl-{uuid.uuid4().hex[:8]}",291 "object": "chat.completion",292 "created": int(time.time()),293 "model": "tokioai-engine",294 "choices": [{295 "index": 0,296 "message": {297 "role": "assistant",298 "content": clean_model_output(result.text or ""),299 },300 "finish_reason": "stop",301 }],302 "usage": {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0},303 "engine_metadata": {304 "investigation_depth": 0,305 "investigation_log": [],306 "elapsed_s": round(elapsed, 1),307 "shortcut": "pure_text",308 },309 })310 311 # === THE INVESTIGATION LOOP ===312 engine.reset()313 314 investigation_log = [] # Track what we tried315 start_time = time.time()316 final_result = None317 delegated_tool_calls = None318 depth = 0319 320 for depth in range(MAX_INVESTIGATION_DEPTH):321 elapsed = time.time() - start_time322 if elapsed > MAX_INVESTIGATION_TIME:323 # Time limit -- force text summary324 if engine_config.verbose:325 print(f"[ENGINE] Investigation time limit reached ({elapsed:.1f}s)")326 break327 328 # Build the prompt for this round329 if depth == 0 and not is_tool_continuation:330 # First round: use original query331 current_msg = user_msg332 force_intent = None333 elif depth == 0 and is_tool_continuation:334 # CLI sent back tool results -- summarize and continue335 current_msg = _build_continuation_prompt(messages, original_query)336 force_intent = None337 else:338 # Investigation round N > 0: tell model what happened339 current_msg = _build_investigation_prompt(340 original_query, investigation_log, depth341 )342 # For simple queries (1 success is enough), let model decide343 # For ongoing investigation, force tool calls until near the limit344 successes_so_far = sum(1 for e in investigation_log if e["success"])345 if depth >= MAX_INVESTIGATION_DEPTH - 2:346 force_intent = "text" # Near limit: force final answer347 elif successes_so_far >= 1 and depth <= 2:348 force_intent = None # Let model decide (may answer or investigate more)349 else:350 force_intent = None # Let model decide351 352 # Call the model353 try:354 result = engine.process(355 user_message=current_msg,356 conversation_history=working_messages,357 force_intent=force_intent,358 )359 except Exception as e:360 if engine_config.verbose:361 print(f"[ENGINE] Model error at depth {depth}: {e}")362 final_result = PipelineResult(363 output_type="text",364 text=f"Error during investigation: {e}",365 )366 break367 368 if engine_config.verbose:369 if result.tool_calls:370 names = [tc["name"] for tc in result.tool_calls]371 print(f"[ENGINE] Depth {depth}: tool_calls={names}")372 else:373 print(f"[ENGINE] Depth {depth}: text response ({len(result.text)} chars)")374 375 # === MODEL RETURNED TEXT ===376 if result.output_type == "text" or not result.tool_calls:377 raw_text = result.text or ""378 text = raw_text.lower()379 380 # CRITICAL: Detect FAKE tool execution (hallucinated command output)381 if has_fake_tool_execution(raw_text) and depth < MAX_INVESTIGATION_DEPTH - 1:382 if engine_config.verbose:383 print(f"[ENGINE] FAKE tool execution detected at depth {depth}, forcing real tool call")384 working_messages.append({385 "role": "user",386 "content": (387 f"STOP. You just simulated a command in text instead of actually "388 f"running it. That output is FAKE. Use a real tool_call to execute "389 f"the command. Call execute_local with the appropriate command. "390 f"Question: {original_query}"391 ),392 })393 continue # Force another round with real tool394 395 # Detect LAZY/INCOMPLETE answers and force more investigation396 lazy_patterns = [397 "you should", "te recomiendo", "se recomienda",398 "requires further", "requiere.*más profund",399 "necesitaría", "se necesitaría", "se necesita",400 "want me to check", "quieres que revise",401 "te gustaría que", "would you like me to",402 "para evaluar", "para determinar",403 "no se detectaron vulnerabilidades inmediatas",404 "you could check", "podrias revisar",405 "voy a verificar", "voy a ejecutar",406 "ejecutemos", "let me run", "let me check",407 ]408 is_lazy = any(p in text for p in lazy_patterns)409 successes_so_far = sum(1 for e in investigation_log if e["success"])410 411 if is_lazy and successes_so_far < 4 and depth < MAX_INVESTIGATION_DEPTH - 2:412 if engine_config.verbose:413 print(f"[ENGINE] Lazy answer detected at depth {depth}, pushing for more investigation")414 working_messages.append({"role": "assistant", "content": raw_text})415 working_messages.append({416 "role": "user",417 "content": (418 f"Your answer was incomplete. You said things like 'you should check' "419 f"or 'I will verify'. YOU are the expert -- don't describe what you "420 f"will do, ACTUALLY DO IT with a tool call right now. "421 f"Original question: {original_query}"422 ),423 })424 continue # Force another round425 426 # Clean the output before accepting it427 result.text = clean_model_output(raw_text)428 final_result = result429 break430 431 # === MODEL RETURNED TOOL CALLS ===432 # Check if all tools are safe for server-side execution433 all_safe = all(is_safe_tool(tc["name"]) for tc in result.tool_calls)434 435 if not all_safe:436 # At least one tool needs CLI execution (write/remote)437 # Return tool_calls to CLI, investigation pauses438 delegated_tool_calls = result439 investigation_stats["delegated_to_cli"] += 1440 if engine_config.verbose:441 unsafe = [tc["name"] for tc in result.tool_calls if not is_safe_tool(tc["name"])]442 print(f"[ENGINE] Delegating to CLI: {unsafe}")443 break444 445 # ALL tools are safe -- execute server-side!446 tool_results = []447 for tc in result.tool_calls:448 name = tc["name"]449 args = tc["arguments"]450 451 if engine_config.verbose:452 arg_preview = args.get("command", args.get("path", args.get("pattern", args.get("target", str(args)))))453 print(f"[ENGINE] Executing: {name}({arg_preview})")454 455 success, output = execute_tool(name, args, timeout=30)456 457 tool_results.append({458 "name": name,459 "arguments": args,460 "output": output,461 "success": success,462 })463 464 investigation_log.append({465 "name": name,466 "arguments": args,467 "output": output[:500], # Keep log manageable468 "success": success,469 "depth": depth,470 })471 472 if engine_config.verbose:473 status = "OK" if success else "FAIL"474 print(f"[ENGINE] Result [{status}]: {output[:100]}")475 476 # Feed tool results back to model using Ollama's native tool format477 # Ollama /api/chat expects: arguments as dict (not JSON string),478 # no "id"/"type" in tool_calls, no "tool_call_id" in tool messages479 480 # Add assistant message with tool_calls481 working_messages.append({482 "role": "assistant",483 "content": "",484 "tool_calls": [485 {486 "function": {487 "name": tr["name"],488 "arguments": tr["arguments"] if isinstance(tr["arguments"], dict) else json.loads(tr["arguments"]),489 },490 }491 for tr in tool_results492 ],493 })494 495 # Add each tool result as a tool message496 for tr in tool_results:497 working_messages.append({498 "role": "tool",499 "content": tr["output"],500 })501 502 # === END OF INVESTIGATION LOOP ===503 504 # Update stats505 if depth > 0:506 investigation_stats["total_investigations"] += 1507 investigation_stats["total_steps"] += depth508 investigation_stats["max_depth_reached"] = max(509 investigation_stats["max_depth_reached"], depth510 )511 total = investigation_stats["total_investigations"]512 investigation_stats["avg_steps"] = (513 investigation_stats["total_steps"] / total if total > 0 else 0514 )515 516 # If we hit the depth/time limit without a final answer, force one517 if final_result is None and delegated_tool_calls is None:518 # Force the model to summarize519 summary_prompt = _build_summary_prompt(original_query, investigation_log)520 try:521 final_result = engine.process(522 user_message=summary_prompt,523 conversation_history=working_messages,524 force_intent="text",525 )526 except:527 final_result = PipelineResult(528 output_type="text",529 text=_format_investigation_summary(original_query, investigation_log),530 )531 532 # Use delegated tool calls if that's what we have533 if delegated_tool_calls is not None:534 result_to_return = delegated_tool_calls535 else:536 result_to_return = final_result537 538 chat_id = f"chatcmpl-{uuid.uuid4().hex[:8]}"539 created = int(time.time())540 541 # --- STREAMING RESPONSE (SSE) ---542 if stream:543 def generate_sse():544 msg = _build_response_message(result_to_return)545 finish_reason = "tool_calls" if result_to_return.tool_calls else "stop"546 547 if result_to_return.tool_calls:548 tc_deltas = []549 for i, tc in enumerate(result_to_return.tool_calls):550 tc_deltas.append({551 "index": i,552 "id": f"call_{uuid.uuid4().hex[:8]}",553 "type": "function",554 "function": {555 "name": tc["name"],556 "arguments": json.dumps(tc["arguments"]) if isinstance(tc["arguments"], dict) else tc["arguments"],557 },558 })559 chunk = {560 "id": chat_id,561 "object": "chat.completion.chunk",562 "created": created,563 "model": "tokioai-engine",564 "choices": [{565 "index": 0,566 "delta": {567 "role": "assistant",568 "content": None,569 "tool_calls": tc_deltas,570 },571 "finish_reason": finish_reason,572 }],573 }574 yield f"data: {json.dumps(chunk)}\n\n"575 else:576 text = clean_model_output(result_to_return.text or "")577 578 # First chunk: role579 chunk = {580 "id": chat_id,581 "object": "chat.completion.chunk",582 "created": created,583 "model": "tokioai-engine",584 "choices": [{585 "index": 0,586 "delta": {"role": "assistant", "content": ""},587 "finish_reason": None,588 }],589 }590 yield f"data: {json.dumps(chunk)}\n\n"591 592 if text:593 # Stream word by word594 words = text.split(' ')595 for i, word in enumerate(words):596 token = word if i == 0 else ' ' + word597 chunk = {598 "id": chat_id,599 "object": "chat.completion.chunk",600 "created": created,601 "model": "tokioai-engine",602 "choices": [{603 "index": 0,604 "delta": {"content": token},605 "finish_reason": None,606 }],607 }608 yield f"data: {json.dumps(chunk)}\n\n"609 610 # Final chunk611 chunk = {612 "id": chat_id,613 "object": "chat.completion.chunk",614 "created": created,615 "model": "tokioai-engine",616 "choices": [{617 "index": 0,618 "delta": {},619 "finish_reason": "stop",620 }],621 }622 yield f"data: {json.dumps(chunk)}\n\n"623 624 yield "data: [DONE]\n\n"625 626 return Response(627 generate_sse(),628 mimetype="text/event-stream",629 headers={630 "Cache-Control": "no-cache",631 "X-Accel-Buffering": "no",632 },633 )634 635 # --- NON-STREAMING RESPONSE ---636 response = {637 "id": chat_id,638 "object": "chat.completion",639 "created": created,640 "model": "tokioai-engine",641 "choices": [642 {643 "index": 0,644 "message": _build_response_message(result_to_return),645 "finish_reason": "tool_calls" if result_to_return.tool_calls else "stop",646 }647 ],648 "usage": {649 "prompt_tokens": 0,650 "completion_tokens": 0,651 "total_tokens": 0,652 },653 "engine_metadata": {654 "investigation_depth": depth,655 "investigation_log": [656 {"tool": e["name"], "success": e["success"]}657 for e in investigation_log658 ],659 "elapsed_s": round(time.time() - start_time, 1),660 },661 }662 663 return jsonify(response)664 665 # ============================================================666 # HELPER FUNCTIONS667 # ============================================================668 669 def _extract_messages(messages: List[Dict]) -> Tuple[str, List[Dict]]:670 """671 Extract the user message and build Ollama-compatible history.672 Converts tool messages to assistant context.673 Returns (user_msg, ollama_messages).674 """675 user_msg = None676 history = []677 678 for msg in messages:679 role = msg.get("role", "")680 content = msg.get("content", "")681 682 if role == "system":683 history.append({"role": "system", "content": content})684 685 elif role == "user":686 if user_msg is not None:687 history.append({"role": "user", "content": user_msg})688 user_msg = content689 690 elif role == "assistant":691 tool_calls = msg.get("tool_calls", [])692 if tool_calls:693 # Convert OpenAI tool_calls format to Ollama format694 # Ollama wants: arguments as dict, no id/type fields695 ollama_tcs = []696 for tc in tool_calls:697 func = tc.get("function", {})698 name = func.get("name", "")699 raw_args = func.get("arguments", {})700 if isinstance(raw_args, str):701 try:702 raw_args = json.loads(raw_args)703 except (json.JSONDecodeError, TypeError):704 raw_args = {}705 ollama_tcs.append({706 "function": {"name": name, "arguments": raw_args},707 })708 history.append({709 "role": "assistant",710 "content": content or "",711 "tool_calls": ollama_tcs,712 })713 elif content:714 history.append({"role": "assistant", "content": content})715 716 elif role == "tool":717 # Ollama tool messages: just role + content (no tool_call_id)718 history.append({719 "role": "tool",720 "content": content or "(no output)",721 })722 723 return user_msg, history724 725 def _build_continuation_prompt(messages: List[Dict], original_query: str) -> str:726 """Build prompt when CLI sends back results from delegated tools."""727 # Find the last tool result728 last_tool_result = ""729 last_tool_name = ""730 for msg in reversed(messages):731 if msg.get("role") == "tool":732 last_tool_result = msg.get("content", "")[:500]733 last_tool_name = msg.get("name", "tool")734 break735 736 return (737 f"The tool {last_tool_name} returned:\n{last_tool_result}\n\n"738 f"Original question: {original_query}\n\n"739 f"Based on this result, either:\n"740 f"1. Call more tools to continue investigating, OR\n"741 f"2. Provide your final answer if you have enough information."742 )743 744 def _build_investigation_prompt(745 original_query: str,746 investigation_log: List[Dict],747 depth: int,748 ) -> str:749 """750 Build a prompt for the next investigation round.751 752 v4.0 PHILOSOPHY: Let the model decide when it has enough data.753 Don't force early answers. Only push for final answer near depth limit.754 """755 failures = sum(1 for e in investigation_log if not e["success"])756 successes = sum(1 for e in investigation_log if e["success"])757 758 # Near the depth limit -- force final answer759 if depth >= MAX_INVESTIGATION_DEPTH - 2:760 return (761 f"FINAL ROUND. You have gathered data from {successes} successful "762 f"tool calls. Provide your FINAL answer to: {original_query}\n\n"763 f"Summarize ALL findings. Be specific -- report exact values, "764 f"ports, processes, permissions. No more tool calls."765 )766 767 # All failures so far -- try different approach768 if failures > 0 and successes == 0:769 return (770 f"The previous commands failed. Try a DIFFERENT approach "771 f"to answer: {original_query}\n"772 f"Use a different command. Do NOT repeat what already failed."773 )774 775 # Normal continuation -- let model freely choose776 already = [e["arguments"].get("command", e["arguments"].get("path", "?"))[:50] for e in investigation_log[-3:]]777 return (778 f"Original question: {original_query}\n"779 f"You've run {len(investigation_log)} commands ({successes} OK, {failures} failed).\n"780 f"Recent: {', '.join(already)}\n\n"781 f"If you have enough information, provide your FINAL answer in plain text.\n"782 f"If you need more data, call another tool with a DIFFERENT command."783 )784 785 def _build_summary_prompt(original_query: str, investigation_log: List[Dict]) -> str:786 """Build a forced-text prompt summarizing all investigation findings."""787 return (788 f"Based on all the tool results above, provide a CLEAR, DEFINITIVE "789 f"answer to: {original_query}\n\n"790 f"Rules for your response:\n"791 f"- Write in plain language, no command blocks\n"792 f"- State what IS working, what ISN'T, and why\n"793 f"- Be direct and specific\n"794 f"- Do NOT include any bash/shell commands or simulated tool output"795 )796 797 def _format_investigation_summary(original_query: str, investigation_log: List[Dict]) -> str:798 """Fallback text summary if model fails to summarize."""799 lines = [f"Investigation: {original_query}\n"]800 for entry in investigation_log:801 status = "OK" if entry["success"] else "FAIL"802 arg_str = entry["arguments"].get("command", entry["arguments"].get("path", str(entry["arguments"])))803 lines.append(f" [{status}] {entry['name']}({arg_str})")804 if entry["output"]:805 lines.append(f" -> {entry['output'][:150]}")806 return "\n".join(lines)807 808 def _build_response_message(result: PipelineResult) -> Dict:809 """Build OpenAI-style message from pipeline result."""810 if result.output_type == "tool_call" and result.tool_calls:811 return {812 "role": "assistant",813 "content": None,814 "tool_calls": [815 {816 "id": f"call_{uuid.uuid4().hex[:8]}",817 "type": "function",818 "function": {819 "name": tc["name"],820 "arguments": json.dumps(tc["arguments"]) if isinstance(tc["arguments"], dict) else tc["arguments"],821 },822 }823 for tc in result.tool_calls824 ],825 }826 return {827 "role": "assistant",828 "content": clean_model_output(result.text or ""),829 }830 831 return app832 833 834def main():835 parser = argparse.ArgumentParser(description="TokioAI Engine Server v3.0")836 parser.add_argument("command", choices=["serve"], help="Server command")837 parser.add_argument("--port", type=int, default=8080, help="Server port")838 parser.add_argument("--host", type=str, default="127.0.0.1", help="Bind host")839 parser.add_argument("--ollama-host", type=str, default=None, help="Ollama API host")840 parser.add_argument("--model", type=str, default="tokioai-coder", help="Model name")841 parser.add_argument("--verbose", action="store_true", help="Verbose logging")842 843 args = parser.parse_args()844 845 config = EngineConfig(846 ollama_host=args.ollama_host,847 model_name=args.model,848 verbose=args.verbose,849 )850 851 app = create_app(config)852 print(f"TokioAI Engine v3.0 starting on {args.host}:{args.port}")853 print(f" Model: {config.model_name}")854 print(f" Ollama: {config.ollama_host}")855 print(f" Max investigation depth: {MAX_INVESTIGATION_DEPTH}")856 print(f" Max investigation time: {MAX_INVESTIGATION_TIME}s")857 print(f" Server-side execution: ENABLED (tools: {', '.join(sorted(SAFE_TOOLS))})")858 print(f" Verbose: {args.verbose}")859 860 app.run(host=args.host, port=args.port, threaded=True)861 862 863if __name__ == "__main__":864 main()865 