Team Ai
Modelpublic

MrMoz33/tokioai-coder-iot

sourceHugging Facemitupdated 5d agoView on Hugging Face
0likes
server.py865 linesDownload Raw Back to engine
1"""2TokioAI Engine v3.0 -- OpenAI-Compatible API Server3SERVER-SIDE MULTI-STEP INVESTIGATION (the Opus architecture)4 5The key breakthrough: when the model calls a read-only tool (execute_local,6read_file, search_files, diagnose), the ENGINE executes it directly and feeds7the result back to the model -- all within a SINGLE API request. This means:8 91. User asks "is Home Assistant running?"102. Model calls execute_local(systemctl status home-assistant)113. Engine EXECUTES it -> "not found"124. Engine feeds result back to model135. Model tries execute_local(docker ps | grep -i home)146. Engine EXECUTES it -> "not found"157. Model tries execute_local(ss -tlnp | grep 8123)168. Engine EXECUTES it -> port 8123 listening179. Model tries execute_local(curl -s http://localhost:8123/api/)1810. Engine EXECUTES it -> HA API responds1911. Model answers: "Yes, HA is running on port 8123 via Docker"20 21All 10 steps happen in ONE API call. The CLI sees only the final answer.22For WRITE tools (write_file, edit_file), the Engine delegates to the CLI.23 24Endpoints:25  POST /v1/chat/completions  -- Standard chat completion (with tool support)26  GET  /v1/models            -- List available models27  GET  /health               -- Health check28  GET  /stats                -- Engine statistics29"""30 31import json32import os33import sys34import time35import uuid36import argparse37import re38from typing import Dict, List, Optional, Tuple39 40from .pipeline import Pipeline, EngineConfig, PipelineResult41from .router import classify_intent42from .executor import execute_tool, is_safe_tool, SAFE_TOOLS, DELEGATE_TOOLS43 44 45def has_fake_tool_execution(text: str) -> bool:46    """Detect if text contains hallucinated/fake tool execution.47    48    ANY bash/shell code block with a system command means the model is 49    writing commands in text instead of using real tool_calls. This is 50    ALWAYS wrong -- the model should call execute_local, not print commands.51    """52    if not text:53        return False54    t = text.lower()55    # Pattern: ANY code block with a system command = fake tool execution56    # The model should be using tool_calls, not writing bash blocks57    if re.search(r'```(?:bash|shell|sh)?\s*\n\s*(?:ss|netstat|nmap|curl|ps|docker|find|grep|ls|cat|df|du|systemctl|journalctl|free|top|uptime|uname|id|ifconfig|ip|iptables|who|w|last|lastlog|users|finger|hostname|whoami|date|mount|lsblk|lsof|apt|dpkg|pip|npm|service|timedatectl|hostnamectl|sleep|HA_TOKEN)\b', t):58        return True59    if 'ejecutemos' in t or 'let me run' in t or 'voy a ejecutar' in t or 'voy a verificar' in t:60        if '```' in t:61            return True62    # Pattern: any code block immediately followed by "Result:" with fake data63    if re.search(r'```[^`]*```\s*\n*result\s*:', t):64        return True65    return False66 67 68def clean_model_output(text: str) -> str:69    """Remove model artifacts and hallucinated tool execution from output text."""70    if not text:71        return text72    # Remove trailing <tool_call> or [TOOL_CALLS] artifacts73    text = re.sub(r'</?tool_call>\s*$', '', text)74    text = re.sub(r'\[TOOL_CALLS?\]\s*$', '', text)75    text = re.sub(r'\[/TOOL_CALLS?\]\s*$', '', text)76    # Remove partial JSON tool call attempts at the end77    text = re.sub(r'\{"name"\s*:\s*"[^"]*"\s*,?\s*("arguments".*)?$', '', text)78    # Remove markdown-wrapped empty tool calls79    text = re.sub(r'```json\s*\{[^}]*\}\s*```\s*$', '', text)80    81    # Remove ALL code blocks containing system commands + fake output82    # This catches: ```bash\nss -tuln\n```\n\nResult: ...\n```...```83    text = re.sub(84        r'```(?:bash|shell|sh)?\s*\n\s*(?:ss|netstat|nmap|curl|ps|docker|find|grep|ls|cat|df|du|systemctl|journalctl|apt|pip|npm|which|whereis|locate|dpkg|service|mount|lsblk|lsof|free|top|htop|uptime|hostname|uname|whoami|id|ifconfig|ip|ping|traceroute|dig|nslookup|iptables|sudo|who|w|last|lastlog|users|date|timedatectl)\b[^`]*```',85        '', text, flags=re.MULTILINE | re.DOTALL86    )87    88    # Remove fake "Result:" sections (with or without code blocks)89    text = re.sub(90        r'(?:Result|Output|Salida)\s*:\s*(?:Active|Proto|tcp|udp|LISTEN|Filesystem|total|Mem:|Swap:).*?(?=\n\n[A-Z]|\Z)',91        '', text, flags=re.DOTALL | re.IGNORECASE92    )93    94    # Remove hallucinated inline tool execution blocks95    # Pattern: ```bash\nCOMMAND\n```\n\nResult of tool:\nFAKE_OUTPUT96    text = re.sub(97        r'```(?:bash|shell|sh)\n[^`]+```\s*\n*(?:Result of tool:?\s*\n[^\n]*(?:\n[^\n]*)*?)(?=\n\n|\Z)',98        '', text, flags=re.MULTILINE99    )100    # Pattern: bare ```\nCOMMAND\n``` blocks (no language tag)101    text = re.sub(102        r'```\n(?:(?:ss|netstat|nmap|curl|ps|docker|find|grep|ls|cat|df|du|systemctl|journalctl|apt|pip|npm|which|whereis|locate|dpkg|service|mount|lsblk|lsof|free|top|htop|uptime|hostname|uname|whoami|id|ifconfig|ip|ping|traceroute|dig|nslookup)\b[^`]*)\n```',103        '', text, flags=re.MULTILINE104    )105    # Pattern: "I ran: tool_name(...)\n[OK] Result:\n..."106    text = re.sub(107        r'I ran:\s*\w+\([^)]*\)\s*\n\[(?:OK|FAIL(?:ED)?)\]\s*Result:\n.*?(?=\n\n|\Z)',108        '', text, flags=re.DOTALL109    )110    # Pattern: standalone "Result of tool:" blocks111    text = re.sub(112        r'Result of tool:\s*\n.*?(?=\n\n|\Z)',113        '', text, flags=re.DOTALL114    )115    # Remove trailing lazy/filler phrases116    lazy_tails = [117        r'\n+(?:Let me know|If you\'?d? like|Do you want|Quieres que|Te gustaría|Si necesitas|Para más detalles).*$',118        r'\n+(?:Would you like me to|Want me to|If you want me to|Si quieres que).*$',119    ]120    for pattern in lazy_tails:121        text = re.sub(pattern, '', text, flags=re.IGNORECASE | re.DOTALL)122    123    # Clean up excessive blank lines left behind124    text = re.sub(r'\n{3,}', '\n\n', text)125    return text.strip()126 127 128# Max number of server-side investigation rounds per request129MAX_INVESTIGATION_DEPTH = 8130 131# Max total time for server-side investigation (seconds)132MAX_INVESTIGATION_TIME = 90133 134 135def count_tool_depth(messages: List[Dict]) -> int:136    """Count tool call rounds in messages."""137    depth = 0138    for msg in messages:139        if msg.get("role") == "assistant" and msg.get("tool_calls"):140            depth += 1141    return depth142 143 144def find_original_query(messages: List[Dict]) -> str:145    """Find the first user message (the original question)."""146    for msg in messages:147        if msg.get("role") == "user":148            content = msg.get("content", "")149            if content and not content.startswith("INVESTIGATION"):150                return content151    return ""152 153 154def create_app(config: EngineConfig = None):155    """Create the Flask app with the engine pipeline."""156    157    try:158        from flask import Flask, request, jsonify, Response159    except ImportError:160        print("Flask not installed. Install with: pip install flask")161        sys.exit(1)162 163    app = Flask("tokioai-engine")164    engine_config = config or EngineConfig()165    engine = Pipeline(engine_config)166 167    # Track investigation stats168    investigation_stats = {169        "total_investigations": 0,170        "total_steps": 0,171        "max_depth_reached": 0,172        "delegated_to_cli": 0,173        "avg_steps": 0,174    }175 176    @app.route("/health", methods=["GET"])177    def health():178        return jsonify({179            "status": "ok",180            "engine": "tokioai-engine",181            "version": "3.0.0",182            "model": engine.config.model_name,183            "ollama_host": engine.config.ollama_host,184            "max_investigation_depth": MAX_INVESTIGATION_DEPTH,185            "max_investigation_time": MAX_INVESTIGATION_TIME,186            "server_side_execution": True,187        })188 189    @app.route("/stats", methods=["GET"])190    def stats():191        s = engine.get_stats()192        s["investigation"] = investigation_stats193        return jsonify(s)194 195    @app.route("/v1/models", methods=["GET"])196    def list_models():197        return jsonify({198            "object": "list",199            "data": [200                {201                    "id": "tokioai-engine",202                    "object": "model",203                    "created": int(time.time()),204                    "owned_by": "tokioai",205                    "permission": [],206                    "root": engine.config.model_name,207                    "parent": None,208                }209            ]210        })211 212    @app.route("/v1/chat/completions", methods=["POST"])213    def chat_completions():214        data = request.get_json()215 216        if not data:217            return jsonify({"error": {"message": "Request body is required"}}), 400218 219        messages = data.get("messages", [])220        if not messages:221            return jsonify({"error": {"message": "messages is required"}}), 400222 223        stream = data.get("stream", False)224 225        # ============================================================226        # v3.0 SERVER-SIDE INVESTIGATION LOOP227        # ============================================================228        #229        # Architecture:230        # 1. Extract user message and conversation history231        # 2. Call model (via Pipeline) -- model returns tool_calls or text232        # 3. If tool_calls:233        #    a. If ALL tools are SAFE (read-only): execute server-side,234        #       feed results back to model, goto step 2235        #    b. If ANY tool is DELEGATE (write/remote): return tool_calls236        #       to CLI for execution (old behavior)237        # 4. If text: return the text answer238        # 5. Repeat up to MAX_INVESTIGATION_DEPTH or MAX_INVESTIGATION_TIME239        240        # Determine if this is a tool-result continuation from CLI241        last_msg_role = None242        for m in reversed(messages):243            if m.get("role") != "system":244                last_msg_role = m.get("role")245                break246        247        is_tool_continuation = (last_msg_role == "tool")248        249        # Build the working message history250        user_msg, working_messages = _extract_messages(messages)251        252        if not user_msg:253            return jsonify({"error": {"message": "No user message found"}}), 400254        255        original_query = user_msg256        257        # If this is a CLI tool continuation (CLI executed a write/remote tool),258        # let the model see the result and decide next step259        if is_tool_continuation:260            # The CLI sent back results from a delegated tool.261            # Feed it all through and let the model continue.262            pass263        264        # === PRE-CLASSIFICATION: Skip investigation for pure text ===265        # For greetings, thanks -- skip the investigation loop entirely.266        # BUT NOT for affirmations ("dale", "si", "ok") which are contextual267        # and need conversation history to understand what the user is agreeing to.268        _GREETING_PATTERNS = [269            r'^(hi|hello|hey|hola|que\s+tal|buenos?\s+d[ií]as?|buenas?\s+(?:tardes?|noches?))\s*[.!?]*$',270            r'^(thanks?|thank\s+you|gracias|thx|de\s+nada)\s*[.!?]*$',271            r'^(como\s+(va|estas?|andas?|te\s+va)|how\s+are\s+you|how\'?s\s+it\s+going|what\'?s\s+up)\s*[.!?]*$',272            r'^(todo\s+bien|bien\s+y\s+tu|genial|cool|nice|great|awesome)\s*[.!?]*$',273            r'^(que\s+onda|que\s+hay|que\s+pasa|que\s+cuentas)\s*[.!?]*$',274            r'\b(who\s+are\s+you|what\s+can\s+you|tell\s+me\s+about\s+yourself)\b',275        ]276        import re as _re277        _is_pure_greeting = any(_re.match(p, user_msg.lower().strip(), _re.IGNORECASE) for p in _GREETING_PATTERNS)278        if not is_tool_continuation and _is_pure_greeting:279                _shortcut_start = time.time()280                if engine_config.verbose:281                    print(f"[ENGINE] Pure text shortcut: '{user_msg}' -> greeting/thanks")282                engine.reset()283                result = engine.process(284                    user_message=user_msg,285                    conversation_history=working_messages,286                    force_intent="text",287                )288                elapsed = time.time() - _shortcut_start289                return jsonify({290                    "id": f"chatcmpl-{uuid.uuid4().hex[:8]}",291                    "object": "chat.completion",292                    "created": int(time.time()),293                    "model": "tokioai-engine",294                    "choices": [{295                        "index": 0,296                        "message": {297                            "role": "assistant",298                            "content": clean_model_output(result.text or ""),299                        },300                        "finish_reason": "stop",301                    }],302                    "usage": {"prompt_tokens": 0, "completion_tokens": 0, "total_tokens": 0},303                    "engine_metadata": {304                        "investigation_depth": 0,305                        "investigation_log": [],306                        "elapsed_s": round(elapsed, 1),307                        "shortcut": "pure_text",308                    },309                })310 311        # === THE INVESTIGATION LOOP ===312        engine.reset()313        314        investigation_log = []  # Track what we tried315        start_time = time.time()316        final_result = None317        delegated_tool_calls = None318        depth = 0319        320        for depth in range(MAX_INVESTIGATION_DEPTH):321            elapsed = time.time() - start_time322            if elapsed > MAX_INVESTIGATION_TIME:323                # Time limit -- force text summary324                if engine_config.verbose:325                    print(f"[ENGINE] Investigation time limit reached ({elapsed:.1f}s)")326                break327            328            # Build the prompt for this round329            if depth == 0 and not is_tool_continuation:330                # First round: use original query331                current_msg = user_msg332                force_intent = None333            elif depth == 0 and is_tool_continuation:334                # CLI sent back tool results -- summarize and continue335                current_msg = _build_continuation_prompt(messages, original_query)336                force_intent = None337            else:338                # Investigation round N > 0: tell model what happened339                current_msg = _build_investigation_prompt(340                    original_query, investigation_log, depth341                )342                # For simple queries (1 success is enough), let model decide343                # For ongoing investigation, force tool calls until near the limit344                successes_so_far = sum(1 for e in investigation_log if e["success"])345                if depth >= MAX_INVESTIGATION_DEPTH - 2:346                    force_intent = "text"  # Near limit: force final answer347                elif successes_so_far >= 1 and depth <= 2:348                    force_intent = None  # Let model decide (may answer or investigate more)349                else:350                    force_intent = None  # Let model decide351            352            # Call the model353            try:354                result = engine.process(355                    user_message=current_msg,356                    conversation_history=working_messages,357                    force_intent=force_intent,358                )359            except Exception as e:360                if engine_config.verbose:361                    print(f"[ENGINE] Model error at depth {depth}: {e}")362                final_result = PipelineResult(363                    output_type="text",364                    text=f"Error during investigation: {e}",365                )366                break367            368            if engine_config.verbose:369                if result.tool_calls:370                    names = [tc["name"] for tc in result.tool_calls]371                    print(f"[ENGINE] Depth {depth}: tool_calls={names}")372                else:373                    print(f"[ENGINE] Depth {depth}: text response ({len(result.text)} chars)")374            375            # === MODEL RETURNED TEXT ===376            if result.output_type == "text" or not result.tool_calls:377                raw_text = result.text or ""378                text = raw_text.lower()379                380                # CRITICAL: Detect FAKE tool execution (hallucinated command output)381                if has_fake_tool_execution(raw_text) and depth < MAX_INVESTIGATION_DEPTH - 1:382                    if engine_config.verbose:383                        print(f"[ENGINE] FAKE tool execution detected at depth {depth}, forcing real tool call")384                    working_messages.append({385                        "role": "user",386                        "content": (387                            f"STOP. You just simulated a command in text instead of actually "388                            f"running it. That output is FAKE. Use a real tool_call to execute "389                            f"the command. Call execute_local with the appropriate command. "390                            f"Question: {original_query}"391                        ),392                    })393                    continue  # Force another round with real tool394                395                # Detect LAZY/INCOMPLETE answers and force more investigation396                lazy_patterns = [397                    "you should", "te recomiendo", "se recomienda",398                    "requires further", "requiere.*más profund",399                    "necesitaría", "se necesitaría", "se necesita",400                    "want me to check", "quieres que revise",401                    "te gustaría que", "would you like me to",402                    "para evaluar", "para determinar",403                    "no se detectaron vulnerabilidades inmediatas",404                    "you could check", "podrias revisar",405                    "voy a verificar", "voy a ejecutar",406                    "ejecutemos", "let me run", "let me check",407                ]408                is_lazy = any(p in text for p in lazy_patterns)409                successes_so_far = sum(1 for e in investigation_log if e["success"])410                411                if is_lazy and successes_so_far < 4 and depth < MAX_INVESTIGATION_DEPTH - 2:412                    if engine_config.verbose:413                        print(f"[ENGINE] Lazy answer detected at depth {depth}, pushing for more investigation")414                    working_messages.append({"role": "assistant", "content": raw_text})415                    working_messages.append({416                        "role": "user",417                        "content": (418                            f"Your answer was incomplete. You said things like 'you should check' "419                            f"or 'I will verify'. YOU are the expert -- don't describe what you "420                            f"will do, ACTUALLY DO IT with a tool call right now. "421                            f"Original question: {original_query}"422                        ),423                    })424                    continue  # Force another round425                426                # Clean the output before accepting it427                result.text = clean_model_output(raw_text)428                final_result = result429                break430            431            # === MODEL RETURNED TOOL CALLS ===432            # Check if all tools are safe for server-side execution433            all_safe = all(is_safe_tool(tc["name"]) for tc in result.tool_calls)434            435            if not all_safe:436                # At least one tool needs CLI execution (write/remote)437                # Return tool_calls to CLI, investigation pauses438                delegated_tool_calls = result439                investigation_stats["delegated_to_cli"] += 1440                if engine_config.verbose:441                    unsafe = [tc["name"] for tc in result.tool_calls if not is_safe_tool(tc["name"])]442                    print(f"[ENGINE] Delegating to CLI: {unsafe}")443                break444            445            # ALL tools are safe -- execute server-side!446            tool_results = []447            for tc in result.tool_calls:448                name = tc["name"]449                args = tc["arguments"]450                451                if engine_config.verbose:452                    arg_preview = args.get("command", args.get("path", args.get("pattern", args.get("target", str(args)))))453                    print(f"[ENGINE]   Executing: {name}({arg_preview})")454                455                success, output = execute_tool(name, args, timeout=30)456                457                tool_results.append({458                    "name": name,459                    "arguments": args,460                    "output": output,461                    "success": success,462                })463                464                investigation_log.append({465                    "name": name,466                    "arguments": args,467                    "output": output[:500],  # Keep log manageable468                    "success": success,469                    "depth": depth,470                })471                472                if engine_config.verbose:473                    status = "OK" if success else "FAIL"474                    print(f"[ENGINE]   Result [{status}]: {output[:100]}")475            476            # Feed tool results back to model using Ollama's native tool format477            # Ollama /api/chat expects: arguments as dict (not JSON string),478            # no "id"/"type" in tool_calls, no "tool_call_id" in tool messages479            480            # Add assistant message with tool_calls481            working_messages.append({482                "role": "assistant",483                "content": "",484                "tool_calls": [485                    {486                        "function": {487                            "name": tr["name"],488                            "arguments": tr["arguments"] if isinstance(tr["arguments"], dict) else json.loads(tr["arguments"]),489                        },490                    }491                    for tr in tool_results492                ],493            })494            495            # Add each tool result as a tool message496            for tr in tool_results:497                working_messages.append({498                    "role": "tool",499                    "content": tr["output"],500                })501        502        # === END OF INVESTIGATION LOOP ===503        504        # Update stats505        if depth > 0:506            investigation_stats["total_investigations"] += 1507            investigation_stats["total_steps"] += depth508            investigation_stats["max_depth_reached"] = max(509                investigation_stats["max_depth_reached"], depth510            )511            total = investigation_stats["total_investigations"]512            investigation_stats["avg_steps"] = (513                investigation_stats["total_steps"] / total if total > 0 else 0514            )515        516        # If we hit the depth/time limit without a final answer, force one517        if final_result is None and delegated_tool_calls is None:518            # Force the model to summarize519            summary_prompt = _build_summary_prompt(original_query, investigation_log)520            try:521                final_result = engine.process(522                    user_message=summary_prompt,523                    conversation_history=working_messages,524                    force_intent="text",525                )526            except:527                final_result = PipelineResult(528                    output_type="text",529                    text=_format_investigation_summary(original_query, investigation_log),530                )531        532        # Use delegated tool calls if that's what we have533        if delegated_tool_calls is not None:534            result_to_return = delegated_tool_calls535        else:536            result_to_return = final_result537        538        chat_id = f"chatcmpl-{uuid.uuid4().hex[:8]}"539        created = int(time.time())540 541        # --- STREAMING RESPONSE (SSE) ---542        if stream:543            def generate_sse():544                msg = _build_response_message(result_to_return)545                finish_reason = "tool_calls" if result_to_return.tool_calls else "stop"546 547                if result_to_return.tool_calls:548                    tc_deltas = []549                    for i, tc in enumerate(result_to_return.tool_calls):550                        tc_deltas.append({551                            "index": i,552                            "id": f"call_{uuid.uuid4().hex[:8]}",553                            "type": "function",554                            "function": {555                                "name": tc["name"],556                                "arguments": json.dumps(tc["arguments"]) if isinstance(tc["arguments"], dict) else tc["arguments"],557                            },558                        })559                    chunk = {560                        "id": chat_id,561                        "object": "chat.completion.chunk",562                        "created": created,563                        "model": "tokioai-engine",564                        "choices": [{565                            "index": 0,566                            "delta": {567                                "role": "assistant",568                                "content": None,569                                "tool_calls": tc_deltas,570                            },571                            "finish_reason": finish_reason,572                        }],573                    }574                    yield f"data: {json.dumps(chunk)}\n\n"575                else:576                    text = clean_model_output(result_to_return.text or "")577 578                    # First chunk: role579                    chunk = {580                        "id": chat_id,581                        "object": "chat.completion.chunk",582                        "created": created,583                        "model": "tokioai-engine",584                        "choices": [{585                            "index": 0,586                            "delta": {"role": "assistant", "content": ""},587                            "finish_reason": None,588                        }],589                    }590                    yield f"data: {json.dumps(chunk)}\n\n"591 592                    if text:593                        # Stream word by word594                        words = text.split(' ')595                        for i, word in enumerate(words):596                            token = word if i == 0 else ' ' + word597                            chunk = {598                                "id": chat_id,599                                "object": "chat.completion.chunk",600                                "created": created,601                                "model": "tokioai-engine",602                                "choices": [{603                                    "index": 0,604                                    "delta": {"content": token},605                                    "finish_reason": None,606                                }],607                            }608                            yield f"data: {json.dumps(chunk)}\n\n"609 610                    # Final chunk611                    chunk = {612                        "id": chat_id,613                        "object": "chat.completion.chunk",614                        "created": created,615                        "model": "tokioai-engine",616                        "choices": [{617                            "index": 0,618                            "delta": {},619                            "finish_reason": "stop",620                        }],621                    }622                    yield f"data: {json.dumps(chunk)}\n\n"623 624                yield "data: [DONE]\n\n"625 626            return Response(627                generate_sse(),628                mimetype="text/event-stream",629                headers={630                    "Cache-Control": "no-cache",631                    "X-Accel-Buffering": "no",632                },633            )634 635        # --- NON-STREAMING RESPONSE ---636        response = {637            "id": chat_id,638            "object": "chat.completion",639            "created": created,640            "model": "tokioai-engine",641            "choices": [642                {643                    "index": 0,644                    "message": _build_response_message(result_to_return),645                    "finish_reason": "tool_calls" if result_to_return.tool_calls else "stop",646                }647            ],648            "usage": {649                "prompt_tokens": 0,650                "completion_tokens": 0,651                "total_tokens": 0,652            },653            "engine_metadata": {654                "investigation_depth": depth,655                "investigation_log": [656                    {"tool": e["name"], "success": e["success"]}657                    for e in investigation_log658                ],659                "elapsed_s": round(time.time() - start_time, 1),660            },661        }662 663        return jsonify(response)664 665    # ============================================================666    # HELPER FUNCTIONS667    # ============================================================668    669    def _extract_messages(messages: List[Dict]) -> Tuple[str, List[Dict]]:670        """671        Extract the user message and build Ollama-compatible history.672        Converts tool messages to assistant context.673        Returns (user_msg, ollama_messages).674        """675        user_msg = None676        history = []677        678        for msg in messages:679            role = msg.get("role", "")680            content = msg.get("content", "")681            682            if role == "system":683                history.append({"role": "system", "content": content})684                685            elif role == "user":686                if user_msg is not None:687                    history.append({"role": "user", "content": user_msg})688                user_msg = content689                690            elif role == "assistant":691                tool_calls = msg.get("tool_calls", [])692                if tool_calls:693                    # Convert OpenAI tool_calls format to Ollama format694                    # Ollama wants: arguments as dict, no id/type fields695                    ollama_tcs = []696                    for tc in tool_calls:697                        func = tc.get("function", {})698                        name = func.get("name", "")699                        raw_args = func.get("arguments", {})700                        if isinstance(raw_args, str):701                            try:702                                raw_args = json.loads(raw_args)703                            except (json.JSONDecodeError, TypeError):704                                raw_args = {}705                        ollama_tcs.append({706                            "function": {"name": name, "arguments": raw_args},707                        })708                    history.append({709                        "role": "assistant",710                        "content": content or "",711                        "tool_calls": ollama_tcs,712                    })713                elif content:714                    history.append({"role": "assistant", "content": content})715                    716            elif role == "tool":717                # Ollama tool messages: just role + content (no tool_call_id)718                history.append({719                    "role": "tool",720                    "content": content or "(no output)",721                })722        723        return user_msg, history724    725    def _build_continuation_prompt(messages: List[Dict], original_query: str) -> str:726        """Build prompt when CLI sends back results from delegated tools."""727        # Find the last tool result728        last_tool_result = ""729        last_tool_name = ""730        for msg in reversed(messages):731            if msg.get("role") == "tool":732                last_tool_result = msg.get("content", "")[:500]733                last_tool_name = msg.get("name", "tool")734                break735        736        return (737            f"The tool {last_tool_name} returned:\n{last_tool_result}\n\n"738            f"Original question: {original_query}\n\n"739            f"Based on this result, either:\n"740            f"1. Call more tools to continue investigating, OR\n"741            f"2. Provide your final answer if you have enough information."742        )743    744    def _build_investigation_prompt(745        original_query: str,746        investigation_log: List[Dict],747        depth: int,748    ) -> str:749        """750        Build a prompt for the next investigation round.751        752        v4.0 PHILOSOPHY: Let the model decide when it has enough data.753        Don't force early answers. Only push for final answer near depth limit.754        """755        failures = sum(1 for e in investigation_log if not e["success"])756        successes = sum(1 for e in investigation_log if e["success"])757        758        # Near the depth limit -- force final answer759        if depth >= MAX_INVESTIGATION_DEPTH - 2:760            return (761                f"FINAL ROUND. You have gathered data from {successes} successful "762                f"tool calls. Provide your FINAL answer to: {original_query}\n\n"763                f"Summarize ALL findings. Be specific -- report exact values, "764                f"ports, processes, permissions. No more tool calls."765            )766        767        # All failures so far -- try different approach768        if failures > 0 and successes == 0:769            return (770                f"The previous commands failed. Try a DIFFERENT approach "771                f"to answer: {original_query}\n"772                f"Use a different command. Do NOT repeat what already failed."773            )774        775        # Normal continuation -- let model freely choose776        already = [e["arguments"].get("command", e["arguments"].get("path", "?"))[:50] for e in investigation_log[-3:]]777        return (778            f"Original question: {original_query}\n"779            f"You've run {len(investigation_log)} commands ({successes} OK, {failures} failed).\n"780            f"Recent: {', '.join(already)}\n\n"781            f"If you have enough information, provide your FINAL answer in plain text.\n"782            f"If you need more data, call another tool with a DIFFERENT command."783        )784    785    def _build_summary_prompt(original_query: str, investigation_log: List[Dict]) -> str:786        """Build a forced-text prompt summarizing all investigation findings."""787        return (788            f"Based on all the tool results above, provide a CLEAR, DEFINITIVE "789            f"answer to: {original_query}\n\n"790            f"Rules for your response:\n"791            f"- Write in plain language, no command blocks\n"792            f"- State what IS working, what ISN'T, and why\n"793            f"- Be direct and specific\n"794            f"- Do NOT include any bash/shell commands or simulated tool output"795        )796    797    def _format_investigation_summary(original_query: str, investigation_log: List[Dict]) -> str:798        """Fallback text summary if model fails to summarize."""799        lines = [f"Investigation: {original_query}\n"]800        for entry in investigation_log:801            status = "OK" if entry["success"] else "FAIL"802            arg_str = entry["arguments"].get("command", entry["arguments"].get("path", str(entry["arguments"])))803            lines.append(f"  [{status}] {entry['name']}({arg_str})")804            if entry["output"]:805                lines.append(f"         -> {entry['output'][:150]}")806        return "\n".join(lines)807 808    def _build_response_message(result: PipelineResult) -> Dict:809        """Build OpenAI-style message from pipeline result."""810        if result.output_type == "tool_call" and result.tool_calls:811            return {812                "role": "assistant",813                "content": None,814                "tool_calls": [815                    {816                        "id": f"call_{uuid.uuid4().hex[:8]}",817                        "type": "function",818                        "function": {819                            "name": tc["name"],820                            "arguments": json.dumps(tc["arguments"]) if isinstance(tc["arguments"], dict) else tc["arguments"],821                        },822                    }823                    for tc in result.tool_calls824                ],825            }826        return {827            "role": "assistant",828            "content": clean_model_output(result.text or ""),829        }830 831    return app832 833 834def main():835    parser = argparse.ArgumentParser(description="TokioAI Engine Server v3.0")836    parser.add_argument("command", choices=["serve"], help="Server command")837    parser.add_argument("--port", type=int, default=8080, help="Server port")838    parser.add_argument("--host", type=str, default="127.0.0.1", help="Bind host")839    parser.add_argument("--ollama-host", type=str, default=None, help="Ollama API host")840    parser.add_argument("--model", type=str, default="tokioai-coder", help="Model name")841    parser.add_argument("--verbose", action="store_true", help="Verbose logging")842 843    args = parser.parse_args()844 845    config = EngineConfig(846        ollama_host=args.ollama_host,847        model_name=args.model,848        verbose=args.verbose,849    )850 851    app = create_app(config)852    print(f"TokioAI Engine v3.0 starting on {args.host}:{args.port}")853    print(f"  Model: {config.model_name}")854    print(f"  Ollama: {config.ollama_host}")855    print(f"  Max investigation depth: {MAX_INVESTIGATION_DEPTH}")856    print(f"  Max investigation time: {MAX_INVESTIGATION_TIME}s")857    print(f"  Server-side execution: ENABLED (tools: {', '.join(sorted(SAFE_TOOLS))})")858    print(f"  Verbose: {args.verbose}")859 860    app.run(host=args.host, port=args.port, threaded=True)861 862 863if __name__ == "__main__":864    main()865