Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
server-test-structured.py1041 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""3Test structured output capability via chat completions endpoint.4 5Each test case contains:6  - response_format: OpenAI-compatible response_format specification.7                     Both "json_schema" and "json_object" are accepted; with8                     "json_object" a schema can be supplied via extra_body.9  - extra_body (optional): dict of extra top-level request fields merged into10                     the request payload (mirrors the OpenAI SDK's extra_body11                     feature; llama.cpp reads a top-level "json_schema" here).12  - messages: initial conversation messages13  - tools (optional): tool definitions (for mixed tool + structured tests)14  - mock_tool_responses (optional): dict mapping tool_name -> callable(arguments) -> str (JSON)15  - apply_stage: "always" to apply response_format to every request,16                 "after_tools" to run the tool loop plain, then request a17                 structured summary in a follow-up user turn.18  - followup (optional, for after_tools): user message appended before the19                 final structured call.20  - validate: callable(parsed_json, tool_calls_history, raw_content) -> (passed: bool, reason: str)21"""22 23import argparse24import json25import requests26import sys27from typing import Any, cast28 29# ---------------------------------------------------------------------------30# Color / formatting helpers31# ---------------------------------------------------------------------------32 33RESET = "\x1b[0m"34BOLD = "\x1b[1m"35DIM = "\x1b[2m"36CYAN = "\x1b[36m"37YELLOW = "\x1b[33m"38GREEN = "\x1b[32m"39RED = "\x1b[31m"40BLUE = "\x1b[34m"41WHITE = "\x1b[97m"42MAGENTA = "\x1b[35m"43 44 45def _print(text="", end="\n"):46    sys.stdout.write(text + end)47    sys.stdout.flush()48 49 50def print_header(title):51    bar = "─" * 6052    _print(f"\n{BOLD}{CYAN}┌{bar}┐{RESET}")53    _print(54        f"{BOLD}{CYAN}│  {WHITE}{title}{CYAN}{' ' * max(0, 58 - len(title))}│{RESET}"55    )56    _print(f"{BOLD}{CYAN}└{bar}┘{RESET}")57 58 59def print_tool_call(name, args):60    args_str = json.dumps(args)61    _print(62        f"\n  {BOLD}{YELLOW}⚙ tool call{RESET}  {CYAN}{name}{RESET}{DIM}({args_str}){RESET}"63    )64 65 66def print_tool_result(result):67    preview = result[:160] + ("…" if len(result) > 160 else "")68    _print(f"  {DIM}{BLUE}↳ result{RESET}    {DIM}{preview}{RESET}")69 70 71def print_model_output(text):72    sys.stdout.write(text)73    sys.stdout.flush()74 75 76def print_pass(reason):77    _print(f"\n{BOLD}{GREEN}✔ PASS{RESET}  {reason}")78 79 80def print_fail(reason):81    _print(f"\n{BOLD}{RED}✘ FAIL{RESET}  {reason}")82 83 84def print_info(msg):85    _print(f"{DIM}{msg}{RESET}")86 87 88def print_schema_note(label, rf, extra_body=None):89    kind = rf.get("type", "?")90    name = ""91    if kind == "json_schema":92        name = rf.get("json_schema", {}).get("name", "")93    elif kind == "json_object" and extra_body and "json_schema" in extra_body:94        extra_schema = extra_body["json_schema"] or {}95        name = extra_schema.get("title") or "extra_body.json_schema"96    _print(f"{DIM}{MAGENTA}  ⟐ response_format [{label}]: {kind}"97           f"{(' / ' + name) if name else ''}{RESET}")98 99 100# ---------------------------------------------------------------------------101# HTTP helpers102# ---------------------------------------------------------------------------103 104 105def chat_completion(url, messages, tools=None, response_format=None, stream=False,106                    extra_body=None):107    payload = {108        "messages": messages,109        "stream": stream,110        "max_tokens": 8192,111    }112    if tools:113        payload["tools"] = tools114        payload["tool_choice"] = "auto"115    if response_format is not None:116        payload["response_format"] = response_format117    if extra_body:118        payload.update(extra_body)119 120    try:121        response = requests.post(url, json=payload, stream=stream)122        response.raise_for_status()123    except requests.exceptions.RequestException as e:124        body = e.response.content if (e.response is not None) else b""125        print_fail(f"Request error: {e} | body: {body}")126        return None127 128    full_content = ""129    reasoning_content = ""130    tool_calls: list[dict] = []131 132    if stream:133        for line in response.iter_lines():134            if not line:135                continue136            decoded = line.decode("utf-8")137            if not decoded.startswith("data: "):138                continue139            data_str = decoded[6:]140            if data_str == "[DONE]":141                break142            try:143                data = json.loads(data_str)144            except json.JSONDecodeError:145                continue146            choices = data.get("choices", [])147            if not choices:148                continue149            delta = choices[0].get("delta", {})150            if delta.get("reasoning_content"):151                reasoning_content += delta["reasoning_content"]152            if delta.get("content"):153                full_content += delta["content"]154                print_model_output(delta["content"])155            for tc in delta.get("tool_calls", []):156                idx = tc.get("index", 0)157                while len(tool_calls) <= idx:158                    tool_calls.append(159                        {160                            "id": "",161                            "type": "function",162                            "function": {"name": "", "arguments": ""},163                        }164                    )165                if "id" in tc:166                    tool_calls[idx]["id"] += tc["id"]167                if "function" in tc:168                    if "name" in tc["function"]:169                        tool_calls[idx]["function"]["name"] += tc["function"]["name"]170                    if "arguments" in tc["function"]:171                        tool_calls[idx]["function"]["arguments"] += tc["function"][172                            "arguments"173                        ]174    else:175        data = response.json()176        choices = data.get("choices", [])177        if choices:178            msg = choices[0].get("message", {})179            full_content = msg.get("content") or ""180            reasoning_content = msg.get("reasoning_content") or ""181            tool_calls = msg.get("tool_calls") or []182            if full_content:183                print_model_output(full_content)184 185    result = {"content": full_content, "tool_calls": tool_calls}186    if reasoning_content:187        result["reasoning_content"] = reasoning_content188    return result189 190 191def run_tool_loop(192    url, messages, tools, mock_tool_responses, stream, response_format=None,193    extra_body=None, max_turns=6,194):195    """196    Drive the tool-call loop. If response_format is provided it is applied to197    every request. Returns (all_tool_calls, final_messages, final_content).198    """199    msgs = list(messages)200    all_tool_calls: list[dict] = []201 202    for _ in range(max_turns):203        result = chat_completion(204            url, msgs, tools=tools, response_format=response_format, stream=stream,205            extra_body=extra_body,206        )207        if result is None:208            return all_tool_calls, msgs, None209 210        tcs = result.get("tool_calls") or []211        content = result.get("content") or ""212 213        if not tcs:214            if content:215                _print(f"\n{DIM}{'·' * 60}{RESET}")216            return all_tool_calls, msgs, content217 218        all_tool_calls.extend(tcs)219 220        assistant_msg: dict = {221            "role": "assistant",222            "content": content,223            "tool_calls": tcs,224        }225        reasoning = result.get("reasoning_content")226        if reasoning:227            assistant_msg["reasoning_content"] = reasoning228        msgs.append(assistant_msg)229 230        for tc in tcs:231            tool_name = tc["function"]["name"]232            try:233                args = json.loads(tc["function"]["arguments"])234            except json.JSONDecodeError:235                args = {}236 237            print_tool_call(tool_name, args)238 239            mock_fn = mock_tool_responses.get(tool_name) if mock_tool_responses else None240            if mock_fn:241                tool_result = mock_fn(args)242            else:243                tool_result = json.dumps({"error": f"Unknown tool: {tool_name}"})244 245            print_tool_result(tool_result)246 247            msgs.append(248                {249                    "role": "tool",250                    "tool_call_id": tc.get("id", ""),251                    "content": tool_result,252                }253            )254 255    return all_tool_calls, msgs, None256 257 258# ---------------------------------------------------------------------------259# Test case runner260# ---------------------------------------------------------------------------261 262 263def _try_parse_json(text):264    """Attempt to parse text as JSON, trimming common markdown fences."""265    if text is None:266        return None267    stripped = text.strip()268    if stripped.startswith("```"):269        lines = stripped.splitlines()270        if lines and lines[0].startswith("```"):271            lines = lines[1:]272        if lines and lines[-1].strip().startswith("```"):273            lines = lines[:-1]274        stripped = "\n".join(lines).strip()275    try:276        return json.loads(stripped)277    except json.JSONDecodeError:278        return None279 280 281def run_test(url, test_case, stream):282    name = test_case["name"]283    mode = f"{'stream' if stream else 'non-stream'}"284    apply_stage = test_case.get("apply_stage", "always")285    print_header(f"{name}  [{mode}] ({apply_stage})")286 287    response_format = test_case["response_format"]288    extra_body = test_case.get("extra_body")289    print_schema_note(apply_stage, response_format, extra_body)290 291    tools = test_case.get("tools")292    mocks = test_case.get("mock_tool_responses") or {}293 294    all_tcs: list[dict] = []295    final_content = None296 297    if apply_stage == "always":298        all_tcs, _msgs, final_content = run_tool_loop(299            url,300            messages=list(test_case["messages"]),301            tools=tools,302            mock_tool_responses=mocks,303            stream=stream,304            response_format=response_format,305            extra_body=extra_body,306        )307    elif apply_stage == "after_tools":308        # Phase 1: plain tool loop, no response_format applied yet.309        all_tcs, msgs, interim_content = run_tool_loop(310            url,311            messages=list(test_case["messages"]),312            tools=tools,313            mock_tool_responses=mocks,314            stream=stream,315            response_format=None,316        )317        if interim_content:318            msgs.append({"role": "assistant", "content": interim_content})319        followup = test_case.get(320            "followup",321            "Now output the answer strictly as JSON matching the provided schema. "322            "Do not include commentary.",323        )324        msgs.append({"role": "user", "content": followup})325 326        # Phase 2: request final structured output. Tools are not passed so the327        # model focuses on producing the schema-constrained answer.328        _print(f"\n{DIM}{MAGENTA}  ⟐ follow-up turn with response_format applied{RESET}")329        result = chat_completion(330            url, msgs, tools=None, response_format=response_format, stream=stream,331            extra_body=extra_body,332        )333        final_content = result["content"] if result else None334    else:335        print_fail(f"Unknown apply_stage: {apply_stage}")336        return False337 338    if final_content is None:339        print_fail("No final content from server.")340        return False341 342    parsed = _try_parse_json(final_content)343    if parsed is None:344        print_fail(f"Final content is not valid JSON: {final_content[:200]!r}")345        return False346 347    passed, reason = test_case["validate"](parsed, all_tcs, final_content)348    if passed:349        print_pass(reason)350    else:351        print_fail(reason)352    return passed353 354 355# ---------------------------------------------------------------------------356# Test case definitions357# ---------------------------------------------------------------------------358 359# ---- Test 1: Book metadata extraction (always / json_schema) ----360 361_BOOK_SCHEMA = {362    "type": "json_schema",363    "json_schema": {364        "name": "book_metadata",365        "strict": True,366        "schema": {367            "type": "object",368            "additionalProperties": False,369            "properties": {370                "title": {"type": "string"},371                "author": {"type": "string"},372                "year": {"type": "integer"},373                "genre": {374                    "type": "string",375                    "enum": [376                        "fiction",377                        "non-fiction",378                        "fantasy",379                        "sci-fi",380                        "mystery",381                        "biography",382                        "history",383                        "other",384                    ],385                },386                "page_count": {"type": "integer"},387            },388            "required": ["title", "author", "year", "genre", "page_count"],389        },390    },391}392 393BOOK_TEST_CASE = {394    "name": "Book metadata extraction (json_schema, always)",395    "response_format": _BOOK_SCHEMA,396    "apply_stage": "always",397    "messages": [398        {399            "role": "user",400            "content": (401                "Extract book metadata from this description: "402                "'Dune is a 1965 science fiction epic by Frank Herbert, spanning roughly "403                "688 pages in its first edition, set on the desert planet Arrakis.' "404                "Return the data as JSON."405            ),406        }407    ],408    "validate": lambda parsed, tcs, raw: _validate_book(parsed),409}410 411 412def _validate_book(parsed):413    required = {"title", "author", "year", "genre", "page_count"}414    missing = required - parsed.keys()415    if missing:416        return False, f"Missing fields: {missing}"417    if not isinstance(parsed["title"], str) or not parsed["title"]:418        return False, "title must be a non-empty string"419    if not isinstance(parsed["author"], str) or "herbert" not in parsed["author"].lower():420        return False, f"author unexpected: {parsed['author']!r}"421    if not isinstance(parsed["year"], int) or parsed["year"] != 1965:422        return False, f"year should be 1965, got {parsed['year']!r}"423    if parsed["genre"] not in {424        "fiction", "non-fiction", "fantasy", "sci-fi", "mystery",425        "biography", "history", "other",426    }:427        return False, f"genre not in enum: {parsed['genre']!r}"428    if not isinstance(parsed["page_count"], int) or parsed["page_count"] <= 0:429        return False, f"page_count should be positive int: {parsed['page_count']!r}"430    return True, f"Book: {parsed['title']} ({parsed['year']}) / {parsed['genre']}"431 432 433# ---- Test 2: Sentiment classification (always / enum-constrained) ----434 435_SENTIMENT_SCHEMA = {436    "type": "json_schema",437    "json_schema": {438        "name": "sentiment_analysis",439        "strict": True,440        "schema": {441            "type": "object",442            "additionalProperties": False,443            "properties": {444                "sentiment": {445                    "type": "string",446                    "enum": ["positive", "negative", "neutral"],447                },448                "confidence": {"type": "number"},449                "keywords": {450                    "type": "array",451                    "items": {"type": "string"},452                    "minItems": 1,453                    "maxItems": 5,454                },455            },456            "required": ["sentiment", "confidence", "keywords"],457        },458    },459}460 461SENTIMENT_TEST_CASE = {462    "name": "Sentiment analysis with enum and array",463    "response_format": _SENTIMENT_SCHEMA,464    "apply_stage": "always",465    "messages": [466        {467            "role": "user",468            "content": (469                "Analyse the sentiment of this review and return JSON with the "470                "detected sentiment label, a confidence score between 0 and 1, "471                "and up to five keyword strings that drove the classification:\n\n"472                "'This product completely exceeded my expectations. The build "473                "quality is phenomenal, it arrived a day early, and customer "474                "support was delightful when I had a setup question.'"475            ),476        }477    ],478    "validate": lambda parsed, tcs, raw: _validate_sentiment(parsed),479}480 481 482def _validate_sentiment(parsed):483    if parsed.get("sentiment") not in {"positive", "negative", "neutral"}:484        return False, f"sentiment not in enum: {parsed.get('sentiment')!r}"485    if parsed["sentiment"] != "positive":486        return False, f"expected positive sentiment, got {parsed['sentiment']}"487    conf = parsed.get("confidence")488    if not isinstance(conf, (int, float)) or not (0.0 <= conf <= 1.0):489        return False, f"confidence not in [0,1]: {conf!r}"490    kws = parsed.get("keywords")491    if not isinstance(kws, list) or not (1 <= len(kws) <= 5):492        return False, f"keywords length out of range: {kws!r}"493    if not all(isinstance(k, str) and k for k in kws):494        return False, f"keywords must be non-empty strings: {kws!r}"495    return True, f"sentiment={parsed['sentiment']} conf={conf} kws={kws}"496 497 498# ---- Test: json_object + extra_body.json_schema (always) ----499#500# Exercises the llama.cpp-specific path where the OpenAI SDK would send501# response_format={"type": "json_object"} and tunnel the schema through502# extra_body.json_schema (which becomes a top-level "json_schema" field on503# the request body).504 505_PRODUCT_JSON_OBJECT_SCHEMA = {506    "$schema": "https://json-schema.org/draft/2020-12/schema",507    "$id": "https://example.com/product.schema.json",508    "title": "Product",509    "description": "A product in the catalog",510    "type": "object",511}512 513PRODUCT_JSON_OBJECT_TEST_CASE = {514    "name": "json_object response_format with extra_body json_schema",515    "response_format": {"type": "json_object"},516    "extra_body": {"json_schema": _PRODUCT_JSON_OBJECT_SCHEMA},517    "apply_stage": "always",518    "messages": [519        {520            "role": "system",521            "content": (522                "Extract structured data from the provided text according to the "523                "JSON schema. Return only valid JSON matching the schema exactly."524            ),525        },526        {527            "role": "user",528            "content": "Product: Wireless Headphones, ID: 101, In Stock: Yes",529        },530    ],531    "validate": lambda parsed, tcs, raw: _validate_product_json_object(parsed),532}533 534 535def _validate_product_json_object(parsed):536    if not isinstance(parsed, dict):537        return False, f"expected JSON object, got {type(parsed).__name__}: {parsed!r}"538    if not parsed:539        return False, f"expected non-empty object, got {parsed!r}"540    return True, f"product object with {len(parsed)} field(s): {sorted(parsed.keys())}"541 542 543# ---- Test 3: Nested recipe schema (always) ----544 545_RECIPE_SCHEMA = {546    "type": "json_schema",547    "json_schema": {548        "name": "recipe",549        "strict": True,550        "schema": {551            "type": "object",552            "additionalProperties": False,553            "properties": {554                "name": {"type": "string"},555                "servings": {"type": "integer"},556                "ingredients": {557                    "type": "array",558                    "minItems": 2,559                    "items": {560                        "type": "object",561                        "additionalProperties": False,562                        "properties": {563                            "item": {"type": "string"},564                            "quantity": {"type": "string"},565                        },566                        "required": ["item", "quantity"],567                    },568                },569                "steps": {570                    "type": "array",571                    "minItems": 2,572                    "items": {"type": "string"},573                },574                "prep_time_minutes": {"type": "integer"},575            },576            "required": ["name", "servings", "ingredients", "steps", "prep_time_minutes"],577        },578    },579}580 581RECIPE_TEST_CASE = {582    "name": "Nested recipe with arrays of objects",583    "response_format": _RECIPE_SCHEMA,584    "apply_stage": "always",585    "messages": [586        {587            "role": "user",588            "content": (589                "Give me a simple 4-serving scrambled eggs recipe as structured JSON. "590                "Include the recipe name, servings, ingredients (each with item and "591                "quantity), preparation steps, and total prep time in minutes."592            ),593        }594    ],595    "validate": lambda parsed, tcs, raw: _validate_recipe(parsed),596}597 598 599def _validate_recipe(parsed):600    required = {"name", "servings", "ingredients", "steps", "prep_time_minutes"}601    missing = required - parsed.keys()602    if missing:603        return False, f"Missing fields: {missing}"604    if not isinstance(parsed["name"], str) or not parsed["name"]:605        return False, "name must be a non-empty string"606    if not isinstance(parsed["servings"], int) or parsed["servings"] <= 0:607        return False, f"servings must be positive int: {parsed['servings']!r}"608    ings = parsed["ingredients"]609    if not isinstance(ings, list) or len(ings) < 2:610        return False, f"ingredients must be array of >=2: got {ings!r}"611    for i, ing in enumerate(ings):612        if not isinstance(ing, dict):613            return False, f"ingredient[{i}] is not an object: {ing!r}"614        ing_d = cast(dict[str, Any], ing)615        item_val = ing_d.get("item")616        qty_val = ing_d.get("quantity")617        if item_val is None or qty_val is None:618            return False, f"ingredient[{i}] missing item/quantity: {ing!r}"619        if not isinstance(item_val, str) or not isinstance(qty_val, str):620            return False, f"ingredient[{i}] fields must be strings: {ing!r}"621    steps = parsed["steps"]622    if not isinstance(steps, list) or len(steps) < 2:623        return False, f"steps must be array of >=2 strings: got {steps!r}"624    if not all(isinstance(s, str) and s for s in steps):625        return False, "all steps must be non-empty strings"626    pt = parsed["prep_time_minutes"]627    if not isinstance(pt, int) or pt <= 0:628        return False, f"prep_time_minutes must be positive int: {pt!r}"629    return True, f"recipe '{parsed['name']}' with {len(ings)} ingredients, {len(steps)} steps"630 631 632# ---- Test 4: Tool call -> structured product comparison (after_tools) ----633 634_SHOP_TOOLS = [635    {636        "type": "function",637        "function": {638            "name": "search_products",639            "description": "Search a product catalogue by keyword.",640            "parameters": {641                "type": "object",642                "properties": {643                    "query": {"type": "string"},644                },645                "required": ["query"],646            },647        },648    },649    {650        "type": "function",651        "function": {652            "name": "get_product_details",653            "description": "Get detailed specs for a product by ID.",654            "parameters": {655                "type": "object",656                "properties": {657                    "product_id": {"type": "string"},658                },659                "required": ["product_id"],660            },661        },662    },663]664 665_SHOP_SEARCH_RESULT = {666    "results": [667        {"product_id": "LAP-001", "title": "AeroBook 13 Pro",       "price": 1399.0, "rating": 4.7},668        {"product_id": "LAP-002", "title": "QuantumSlim 14",        "price": 1199.0, "rating": 4.4},669        {"product_id": "LAP-003", "title": "NimbusWork Ultra 15",   "price":  999.0, "rating": 4.2},670    ],671}672_SHOP_PRODUCT_DETAILS = {673    "LAP-001": {674        "product_id": "LAP-001",675        "title": "AeroBook 13 Pro",676        "cpu": "M-series 10-core",677        "ram_gb": 16,678        "storage_gb": 512,679        "battery_hours": 18,680        "weight_kg": 1.24,681        "price": 1399.0,682    },683    "LAP-002": {684        "product_id": "LAP-002",685        "title": "QuantumSlim 14",686        "cpu": "Core i7 12-core",687        "ram_gb": 16,688        "storage_gb": 512,689        "battery_hours": 12,690        "weight_kg": 1.35,691        "price": 1199.0,692    },693    "LAP-003": {694        "product_id": "LAP-003",695        "title": "NimbusWork Ultra 15",696        "cpu": "Ryzen 7 8-core",697        "ram_gb": 16,698        "storage_gb": 1024,699        "battery_hours": 10,700        "weight_kg": 1.70,701        "price": 999.0,702    },703}704 705 706def _shop_details_mock(args):707    pid = args.get("product_id", "")708    if pid in _SHOP_PRODUCT_DETAILS:709        return json.dumps(_SHOP_PRODUCT_DETAILS[pid])710    return json.dumps({"error": f"unknown product_id: {pid}"})711 712 713_SHOP_COMPARISON_SCHEMA = {714    "type": "json_schema",715    "json_schema": {716        "name": "laptop_comparison",717        "strict": True,718        "schema": {719            "type": "object",720            "additionalProperties": False,721            "properties": {722                "recommendation": {"type": "string"},723                "ranked_candidates": {724                    "type": "array",725                    "minItems": 2,726                    "items": {727                        "type": "object",728                        "additionalProperties": False,729                        "properties": {730                            "product_id": {"type": "string"},731                            "title":      {"type": "string"},732                            "score":      {"type": "number"},733                            "reason":     {"type": "string"},734                        },735                        "required": ["product_id", "title", "score", "reason"],736                    },737                },738            },739            "required": ["recommendation", "ranked_candidates"],740        },741    },742}743 744SHOP_COMPARISON_TEST_CASE = {745    "name": "Tool calls then structured laptop comparison (after_tools)",746    "response_format": _SHOP_COMPARISON_SCHEMA,747    "apply_stage": "after_tools",748    "tools": _SHOP_TOOLS,749    "mock_tool_responses": {750        "search_products": lambda _: json.dumps(_SHOP_SEARCH_RESULT),751        "get_product_details": _shop_details_mock,752    },753    "messages": [754        {755            "role": "user",756            "content": (757                "I need a lightweight laptop for travel. Please search the catalogue "758                "for 'ultraportable laptop', then fetch detailed specs for at least two "759                "of the top candidates. Once you've gathered the data I'll ask you to "760                "produce a structured comparison."761            ),762        }763    ],764    "followup": (765        "Thanks. Now produce the final comparison strictly as JSON matching the "766        "laptop_comparison schema: your single best recommendation (the product_id), "767        "and a ranked_candidates array of at least two laptops, each with "768        "product_id, title, a numeric score, and a short reason."769    ),770    "validate": lambda parsed, tcs, raw: _validate_shop_comparison(parsed, tcs),771}772 773 774def _validate_shop_comparison(parsed, tcs):775    names = [tc["function"]["name"] for tc in tcs]776    if "search_products" not in names:777        return False, f"expected search_products tool call, got {names}"778    if "get_product_details" not in names:779        return False, f"expected get_product_details tool call, got {names}"780    if "recommendation" not in parsed or not isinstance(parsed["recommendation"], str):781        return False, f"recommendation missing or not a string: {parsed!r}"782    cands = parsed.get("ranked_candidates")783    if not isinstance(cands, list) or len(cands) < 2:784        return False, f"ranked_candidates must be >=2: {cands!r}"785    valid_ids = set(_SHOP_PRODUCT_DETAILS.keys())786    candidate_pids: list = []787    for i, c in enumerate(cands):788        if not isinstance(c, dict):789            return False, f"candidate[{i}] not an object: {c!r}"790        c_d = cast(dict[str, Any], c)791        pid = c_d.get("product_id")792        title = c_d.get("title")793        score = c_d.get("score")794        reason = c_d.get("reason")795        for k, v in (("product_id", pid), ("title", title),796                     ("score", score), ("reason", reason)):797            if v is None:798                return False, f"candidate[{i}] missing {k}: {c!r}"799        if pid not in valid_ids:800            return False, f"candidate[{i}].product_id not in catalogue: {pid!r}"801        if not isinstance(score, (int, float)):802            return False, f"candidate[{i}].score not numeric: {score!r}"803        candidate_pids.append(pid)804    recommendation = parsed["recommendation"]805    if recommendation not in valid_ids and recommendation not in candidate_pids:806        return False, f"recommendation {recommendation!r} not in candidates"807    return True, (808        f"tools={names}; recommended={parsed['recommendation']}; "809        f"{len(cands)} ranked candidates"810    )811 812 813# ---- Test 5: Multi-step research then structured report (after_tools) ----814 815_RESEARCH_TOOLS = [816    {817        "type": "function",818        "function": {819            "name": "get_country_stats",820            "description": "Fetch basic statistics for a country (population, GDP, capital).",821            "parameters": {822                "type": "object",823                "properties": {824                    "country": {"type": "string"},825                },826                "required": ["country"],827            },828        },829    },830    {831        "type": "function",832        "function": {833            "name": "get_climate_info",834            "description": "Fetch climate information for a country.",835            "parameters": {836                "type": "object",837                "properties": {838                    "country": {"type": "string"},839                },840                "required": ["country"],841            },842        },843    },844]845 846_COUNTRY_STATS = {847    "norway": {848        "country": "Norway",849        "capital": "Oslo",850        "population": 5_480_000,851        "gdp_usd_trillion": 0.48,852        "currency": "NOK",853    }854}855_CLIMATE_INFO = {856    "norway": {857        "country": "Norway",858        "climate_zone": "subarctic / temperate coastal",859        "avg_winter_temp_c": -4.5,860        "avg_summer_temp_c": 16.0,861        "annual_precipitation_mm": 1400,862    }863}864 865 866def _country_stats_mock(args):867    c = args.get("country", "").strip().lower()868    if c in _COUNTRY_STATS:869        return json.dumps(_COUNTRY_STATS[c])870    return json.dumps({"error": f"unknown country: {c}"})871 872 873def _climate_info_mock(args):874    c = args.get("country", "").strip().lower()875    if c in _CLIMATE_INFO:876        return json.dumps(_CLIMATE_INFO[c])877    return json.dumps({"error": f"unknown country: {c}"})878 879 880_RESEARCH_REPORT_SCHEMA = {881    "type": "json_schema",882    "json_schema": {883        "name": "country_report",884        "strict": True,885        "schema": {886            "type": "object",887            "additionalProperties": False,888            "properties": {889                "country": {"type": "string"},890                "capital": {"type": "string"},891                "population": {"type": "integer"},892                "climate_summary": {"type": "string"},893                "highlights": {894                    "type": "array",895                    "minItems": 2,896                    "maxItems": 5,897                    "items": {"type": "string"},898                },899                "suitable_for_tourism": {"type": "boolean"},900            },901            "required": [902                "country", "capital", "population",903                "climate_summary", "highlights", "suitable_for_tourism",904            ],905        },906    },907}908 909COUNTRY_REPORT_TEST_CASE = {910    "name": "Research pipeline then structured country report (after_tools)",911    "response_format": _RESEARCH_REPORT_SCHEMA,912    "apply_stage": "after_tools",913    "tools": _RESEARCH_TOOLS,914    "mock_tool_responses": {915        "get_country_stats": _country_stats_mock,916        "get_climate_info": _climate_info_mock,917    },918    "messages": [919        {920            "role": "user",921            "content": (922                "I'm preparing a short briefing on Norway. Please call the "923                "get_country_stats and get_climate_info tools to gather data "924                "first. Afterwards I'll ask for a structured summary."925            ),926        }927    ],928    "followup": (929        "Based on the tool results, produce the briefing as JSON matching the "930        "country_report schema. Populate every required field and provide between "931        "two and five highlights."932    ),933    "validate": lambda parsed, tcs, raw: _validate_country_report(parsed, tcs),934}935 936 937def _validate_country_report(parsed, tcs):938    names = [tc["function"]["name"] for tc in tcs]939    for required_tool in ("get_country_stats", "get_climate_info"):940        if required_tool not in names:941            return False, f"missing tool call {required_tool!r}: got {names}"942    required = {943        "country", "capital", "population",944        "climate_summary", "highlights", "suitable_for_tourism",945    }946    missing = required - parsed.keys()947    if missing:948        return False, f"missing report fields: {missing}"949    if "norway" not in parsed["country"].lower():950        return False, f"country should reference Norway: {parsed['country']!r}"951    if "oslo" not in parsed["capital"].lower():952        return False, f"capital should be Oslo: {parsed['capital']!r}"953    if not isinstance(parsed["population"], int) or parsed["population"] < 1_000_000:954        return False, f"population implausible: {parsed['population']!r}"955    if not isinstance(parsed["climate_summary"], str) or not parsed["climate_summary"]:956        return False, "climate_summary must be a non-empty string"957    hls = parsed["highlights"]958    if not isinstance(hls, list) or not (2 <= len(hls) <= 5):959        return False, f"highlights length out of range: {hls!r}"960    if not all(isinstance(h, str) and h for h in hls):961        return False, "each highlight must be a non-empty string"962    if not isinstance(parsed["suitable_for_tourism"], bool):963        return False, f"suitable_for_tourism must be bool: {parsed['suitable_for_tourism']!r}"964    return True, (965        f"tools={names}; report for {parsed['country']} "966        f"(pop {parsed['population']}, {len(hls)} highlights)"967    )968 969 970# ---------------------------------------------------------------------------971# All test cases972# ---------------------------------------------------------------------------973 974ALL_TEST_CASES = [975    BOOK_TEST_CASE,976    SENTIMENT_TEST_CASE,977    PRODUCT_JSON_OBJECT_TEST_CASE,978    RECIPE_TEST_CASE,979    SHOP_COMPARISON_TEST_CASE,980    COUNTRY_REPORT_TEST_CASE,981]982 983 984# ---------------------------------------------------------------------------985# Entry point986# ---------------------------------------------------------------------------987 988 989def main():990    parser = argparse.ArgumentParser(991        description="Test llama-server structured-output capability."992    )993    parser.add_argument("--host", default="localhost")994    parser.add_argument("--port", default=8080, type=int)995    parser.add_argument(996        "--no-stream", action="store_true", help="Disable streaming mode tests"997    )998    parser.add_argument(999        "--stream-only", action="store_true", help="Only run streaming mode tests"1000    )1001    parser.add_argument(1002        "--test",1003        help="Run only the test whose name contains this substring (case-insensitive)",1004    )1005    args = parser.parse_args()1006 1007    url = f"http://{args.host}:{args.port}/v1/chat/completions"1008    print_info(f"Testing server at {url}")1009 1010    modes: list[bool] = []1011    if not args.stream_only:1012        modes.append(False)1013    if not args.no_stream:1014        modes.append(True)1015 1016    cases: list[dict] = ALL_TEST_CASES1017    if args.test:1018        name_filter = args.test.lower()1019        cases = [c for c in cases if name_filter in str(c["name"]).lower()]1020        if not cases:1021            print_fail(f"No test cases matched '{args.test}'")1022            sys.exit(1)1023 1024    total = 01025    passed = 01026    for stream in modes:1027        for case in cases:1028            total += 11029            if run_test(url, case, stream=stream):1030                passed += 11031 1032    color = GREEN if passed == total else RED1033    _print(f"\n{BOLD}{color}{'─' * 60}{RESET}")1034    _print(f"{BOLD}{color}  Results: {passed}/{total} passed{RESET}")1035    _print(f"{BOLD}{color}{'─' * 60}{RESET}\n")1036    sys.exit(0 if passed == total else 1)1037 1038 1039if __name__ == "__main__":1040    main()1041 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai