Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#!/usr/bin/env python32"""3Test structured output capability via chat completions endpoint.4 5Each test case contains:6 - response_format: OpenAI-compatible response_format specification.7 Both "json_schema" and "json_object" are accepted; with8 "json_object" a schema can be supplied via extra_body.9 - extra_body (optional): dict of extra top-level request fields merged into10 the request payload (mirrors the OpenAI SDK's extra_body11 feature; llama.cpp reads a top-level "json_schema" here).12 - messages: initial conversation messages13 - tools (optional): tool definitions (for mixed tool + structured tests)14 - mock_tool_responses (optional): dict mapping tool_name -> callable(arguments) -> str (JSON)15 - apply_stage: "always" to apply response_format to every request,16 "after_tools" to run the tool loop plain, then request a17 structured summary in a follow-up user turn.18 - followup (optional, for after_tools): user message appended before the19 final structured call.20 - validate: callable(parsed_json, tool_calls_history, raw_content) -> (passed: bool, reason: str)21"""22 23import argparse24import json25import requests26import sys27from typing import Any, cast28 29# ---------------------------------------------------------------------------30# Color / formatting helpers31# ---------------------------------------------------------------------------32 33RESET = "\x1b[0m"34BOLD = "\x1b[1m"35DIM = "\x1b[2m"36CYAN = "\x1b[36m"37YELLOW = "\x1b[33m"38GREEN = "\x1b[32m"39RED = "\x1b[31m"40BLUE = "\x1b[34m"41WHITE = "\x1b[97m"42MAGENTA = "\x1b[35m"43 44 45def _print(text="", end="\n"):46 sys.stdout.write(text + end)47 sys.stdout.flush()48 49 50def print_header(title):51 bar = "─" * 6052 _print(f"\n{BOLD}{CYAN}┌{bar}┐{RESET}")53 _print(54 f"{BOLD}{CYAN}│ {WHITE}{title}{CYAN}{' ' * max(0, 58 - len(title))}│{RESET}"55 )56 _print(f"{BOLD}{CYAN}└{bar}┘{RESET}")57 58 59def print_tool_call(name, args):60 args_str = json.dumps(args)61 _print(62 f"\n {BOLD}{YELLOW}⚙ tool call{RESET} {CYAN}{name}{RESET}{DIM}({args_str}){RESET}"63 )64 65 66def print_tool_result(result):67 preview = result[:160] + ("…" if len(result) > 160 else "")68 _print(f" {DIM}{BLUE}↳ result{RESET} {DIM}{preview}{RESET}")69 70 71def print_model_output(text):72 sys.stdout.write(text)73 sys.stdout.flush()74 75 76def print_pass(reason):77 _print(f"\n{BOLD}{GREEN}✔ PASS{RESET} {reason}")78 79 80def print_fail(reason):81 _print(f"\n{BOLD}{RED}✘ FAIL{RESET} {reason}")82 83 84def print_info(msg):85 _print(f"{DIM}{msg}{RESET}")86 87 88def print_schema_note(label, rf, extra_body=None):89 kind = rf.get("type", "?")90 name = ""91 if kind == "json_schema":92 name = rf.get("json_schema", {}).get("name", "")93 elif kind == "json_object" and extra_body and "json_schema" in extra_body:94 extra_schema = extra_body["json_schema"] or {}95 name = extra_schema.get("title") or "extra_body.json_schema"96 _print(f"{DIM}{MAGENTA} ⟐ response_format [{label}]: {kind}"97 f"{(' / ' + name) if name else ''}{RESET}")98 99 100# ---------------------------------------------------------------------------101# HTTP helpers102# ---------------------------------------------------------------------------103 104 105def chat_completion(url, messages, tools=None, response_format=None, stream=False,106 extra_body=None):107 payload = {108 "messages": messages,109 "stream": stream,110 "max_tokens": 8192,111 }112 if tools:113 payload["tools"] = tools114 payload["tool_choice"] = "auto"115 if response_format is not None:116 payload["response_format"] = response_format117 if extra_body:118 payload.update(extra_body)119 120 try:121 response = requests.post(url, json=payload, stream=stream)122 response.raise_for_status()123 except requests.exceptions.RequestException as e:124 body = e.response.content if (e.response is not None) else b""125 print_fail(f"Request error: {e} | body: {body}")126 return None127 128 full_content = ""129 reasoning_content = ""130 tool_calls: list[dict] = []131 132 if stream:133 for line in response.iter_lines():134 if not line:135 continue136 decoded = line.decode("utf-8")137 if not decoded.startswith("data: "):138 continue139 data_str = decoded[6:]140 if data_str == "[DONE]":141 break142 try:143 data = json.loads(data_str)144 except json.JSONDecodeError:145 continue146 choices = data.get("choices", [])147 if not choices:148 continue149 delta = choices[0].get("delta", {})150 if delta.get("reasoning_content"):151 reasoning_content += delta["reasoning_content"]152 if delta.get("content"):153 full_content += delta["content"]154 print_model_output(delta["content"])155 for tc in delta.get("tool_calls", []):156 idx = tc.get("index", 0)157 while len(tool_calls) <= idx:158 tool_calls.append(159 {160 "id": "",161 "type": "function",162 "function": {"name": "", "arguments": ""},163 }164 )165 if "id" in tc:166 tool_calls[idx]["id"] += tc["id"]167 if "function" in tc:168 if "name" in tc["function"]:169 tool_calls[idx]["function"]["name"] += tc["function"]["name"]170 if "arguments" in tc["function"]:171 tool_calls[idx]["function"]["arguments"] += tc["function"][172 "arguments"173 ]174 else:175 data = response.json()176 choices = data.get("choices", [])177 if choices:178 msg = choices[0].get("message", {})179 full_content = msg.get("content") or ""180 reasoning_content = msg.get("reasoning_content") or ""181 tool_calls = msg.get("tool_calls") or []182 if full_content:183 print_model_output(full_content)184 185 result = {"content": full_content, "tool_calls": tool_calls}186 if reasoning_content:187 result["reasoning_content"] = reasoning_content188 return result189 190 191def run_tool_loop(192 url, messages, tools, mock_tool_responses, stream, response_format=None,193 extra_body=None, max_turns=6,194):195 """196 Drive the tool-call loop. If response_format is provided it is applied to197 every request. Returns (all_tool_calls, final_messages, final_content).198 """199 msgs = list(messages)200 all_tool_calls: list[dict] = []201 202 for _ in range(max_turns):203 result = chat_completion(204 url, msgs, tools=tools, response_format=response_format, stream=stream,205 extra_body=extra_body,206 )207 if result is None:208 return all_tool_calls, msgs, None209 210 tcs = result.get("tool_calls") or []211 content = result.get("content") or ""212 213 if not tcs:214 if content:215 _print(f"\n{DIM}{'·' * 60}{RESET}")216 return all_tool_calls, msgs, content217 218 all_tool_calls.extend(tcs)219 220 assistant_msg: dict = {221 "role": "assistant",222 "content": content,223 "tool_calls": tcs,224 }225 reasoning = result.get("reasoning_content")226 if reasoning:227 assistant_msg["reasoning_content"] = reasoning228 msgs.append(assistant_msg)229 230 for tc in tcs:231 tool_name = tc["function"]["name"]232 try:233 args = json.loads(tc["function"]["arguments"])234 except json.JSONDecodeError:235 args = {}236 237 print_tool_call(tool_name, args)238 239 mock_fn = mock_tool_responses.get(tool_name) if mock_tool_responses else None240 if mock_fn:241 tool_result = mock_fn(args)242 else:243 tool_result = json.dumps({"error": f"Unknown tool: {tool_name}"})244 245 print_tool_result(tool_result)246 247 msgs.append(248 {249 "role": "tool",250 "tool_call_id": tc.get("id", ""),251 "content": tool_result,252 }253 )254 255 return all_tool_calls, msgs, None256 257 258# ---------------------------------------------------------------------------259# Test case runner260# ---------------------------------------------------------------------------261 262 263def _try_parse_json(text):264 """Attempt to parse text as JSON, trimming common markdown fences."""265 if text is None:266 return None267 stripped = text.strip()268 if stripped.startswith("```"):269 lines = stripped.splitlines()270 if lines and lines[0].startswith("```"):271 lines = lines[1:]272 if lines and lines[-1].strip().startswith("```"):273 lines = lines[:-1]274 stripped = "\n".join(lines).strip()275 try:276 return json.loads(stripped)277 except json.JSONDecodeError:278 return None279 280 281def run_test(url, test_case, stream):282 name = test_case["name"]283 mode = f"{'stream' if stream else 'non-stream'}"284 apply_stage = test_case.get("apply_stage", "always")285 print_header(f"{name} [{mode}] ({apply_stage})")286 287 response_format = test_case["response_format"]288 extra_body = test_case.get("extra_body")289 print_schema_note(apply_stage, response_format, extra_body)290 291 tools = test_case.get("tools")292 mocks = test_case.get("mock_tool_responses") or {}293 294 all_tcs: list[dict] = []295 final_content = None296 297 if apply_stage == "always":298 all_tcs, _msgs, final_content = run_tool_loop(299 url,300 messages=list(test_case["messages"]),301 tools=tools,302 mock_tool_responses=mocks,303 stream=stream,304 response_format=response_format,305 extra_body=extra_body,306 )307 elif apply_stage == "after_tools":308 # Phase 1: plain tool loop, no response_format applied yet.309 all_tcs, msgs, interim_content = run_tool_loop(310 url,311 messages=list(test_case["messages"]),312 tools=tools,313 mock_tool_responses=mocks,314 stream=stream,315 response_format=None,316 )317 if interim_content:318 msgs.append({"role": "assistant", "content": interim_content})319 followup = test_case.get(320 "followup",321 "Now output the answer strictly as JSON matching the provided schema. "322 "Do not include commentary.",323 )324 msgs.append({"role": "user", "content": followup})325 326 # Phase 2: request final structured output. Tools are not passed so the327 # model focuses on producing the schema-constrained answer.328 _print(f"\n{DIM}{MAGENTA} ⟐ follow-up turn with response_format applied{RESET}")329 result = chat_completion(330 url, msgs, tools=None, response_format=response_format, stream=stream,331 extra_body=extra_body,332 )333 final_content = result["content"] if result else None334 else:335 print_fail(f"Unknown apply_stage: {apply_stage}")336 return False337 338 if final_content is None:339 print_fail("No final content from server.")340 return False341 342 parsed = _try_parse_json(final_content)343 if parsed is None:344 print_fail(f"Final content is not valid JSON: {final_content[:200]!r}")345 return False346 347 passed, reason = test_case["validate"](parsed, all_tcs, final_content)348 if passed:349 print_pass(reason)350 else:351 print_fail(reason)352 return passed353 354 355# ---------------------------------------------------------------------------356# Test case definitions357# ---------------------------------------------------------------------------358 359# ---- Test 1: Book metadata extraction (always / json_schema) ----360 361_BOOK_SCHEMA = {362 "type": "json_schema",363 "json_schema": {364 "name": "book_metadata",365 "strict": True,366 "schema": {367 "type": "object",368 "additionalProperties": False,369 "properties": {370 "title": {"type": "string"},371 "author": {"type": "string"},372 "year": {"type": "integer"},373 "genre": {374 "type": "string",375 "enum": [376 "fiction",377 "non-fiction",378 "fantasy",379 "sci-fi",380 "mystery",381 "biography",382 "history",383 "other",384 ],385 },386 "page_count": {"type": "integer"},387 },388 "required": ["title", "author", "year", "genre", "page_count"],389 },390 },391}392 393BOOK_TEST_CASE = {394 "name": "Book metadata extraction (json_schema, always)",395 "response_format": _BOOK_SCHEMA,396 "apply_stage": "always",397 "messages": [398 {399 "role": "user",400 "content": (401 "Extract book metadata from this description: "402 "'Dune is a 1965 science fiction epic by Frank Herbert, spanning roughly "403 "688 pages in its first edition, set on the desert planet Arrakis.' "404 "Return the data as JSON."405 ),406 }407 ],408 "validate": lambda parsed, tcs, raw: _validate_book(parsed),409}410 411 412def _validate_book(parsed):413 required = {"title", "author", "year", "genre", "page_count"}414 missing = required - parsed.keys()415 if missing:416 return False, f"Missing fields: {missing}"417 if not isinstance(parsed["title"], str) or not parsed["title"]:418 return False, "title must be a non-empty string"419 if not isinstance(parsed["author"], str) or "herbert" not in parsed["author"].lower():420 return False, f"author unexpected: {parsed['author']!r}"421 if not isinstance(parsed["year"], int) or parsed["year"] != 1965:422 return False, f"year should be 1965, got {parsed['year']!r}"423 if parsed["genre"] not in {424 "fiction", "non-fiction", "fantasy", "sci-fi", "mystery",425 "biography", "history", "other",426 }:427 return False, f"genre not in enum: {parsed['genre']!r}"428 if not isinstance(parsed["page_count"], int) or parsed["page_count"] <= 0:429 return False, f"page_count should be positive int: {parsed['page_count']!r}"430 return True, f"Book: {parsed['title']} ({parsed['year']}) / {parsed['genre']}"431 432 433# ---- Test 2: Sentiment classification (always / enum-constrained) ----434 435_SENTIMENT_SCHEMA = {436 "type": "json_schema",437 "json_schema": {438 "name": "sentiment_analysis",439 "strict": True,440 "schema": {441 "type": "object",442 "additionalProperties": False,443 "properties": {444 "sentiment": {445 "type": "string",446 "enum": ["positive", "negative", "neutral"],447 },448 "confidence": {"type": "number"},449 "keywords": {450 "type": "array",451 "items": {"type": "string"},452 "minItems": 1,453 "maxItems": 5,454 },455 },456 "required": ["sentiment", "confidence", "keywords"],457 },458 },459}460 461SENTIMENT_TEST_CASE = {462 "name": "Sentiment analysis with enum and array",463 "response_format": _SENTIMENT_SCHEMA,464 "apply_stage": "always",465 "messages": [466 {467 "role": "user",468 "content": (469 "Analyse the sentiment of this review and return JSON with the "470 "detected sentiment label, a confidence score between 0 and 1, "471 "and up to five keyword strings that drove the classification:\n\n"472 "'This product completely exceeded my expectations. The build "473 "quality is phenomenal, it arrived a day early, and customer "474 "support was delightful when I had a setup question.'"475 ),476 }477 ],478 "validate": lambda parsed, tcs, raw: _validate_sentiment(parsed),479}480 481 482def _validate_sentiment(parsed):483 if parsed.get("sentiment") not in {"positive", "negative", "neutral"}:484 return False, f"sentiment not in enum: {parsed.get('sentiment')!r}"485 if parsed["sentiment"] != "positive":486 return False, f"expected positive sentiment, got {parsed['sentiment']}"487 conf = parsed.get("confidence")488 if not isinstance(conf, (int, float)) or not (0.0 <= conf <= 1.0):489 return False, f"confidence not in [0,1]: {conf!r}"490 kws = parsed.get("keywords")491 if not isinstance(kws, list) or not (1 <= len(kws) <= 5):492 return False, f"keywords length out of range: {kws!r}"493 if not all(isinstance(k, str) and k for k in kws):494 return False, f"keywords must be non-empty strings: {kws!r}"495 return True, f"sentiment={parsed['sentiment']} conf={conf} kws={kws}"496 497 498# ---- Test: json_object + extra_body.json_schema (always) ----499#500# Exercises the llama.cpp-specific path where the OpenAI SDK would send501# response_format={"type": "json_object"} and tunnel the schema through502# extra_body.json_schema (which becomes a top-level "json_schema" field on503# the request body).504 505_PRODUCT_JSON_OBJECT_SCHEMA = {506 "$schema": "https://json-schema.org/draft/2020-12/schema",507 "$id": "https://example.com/product.schema.json",508 "title": "Product",509 "description": "A product in the catalog",510 "type": "object",511}512 513PRODUCT_JSON_OBJECT_TEST_CASE = {514 "name": "json_object response_format with extra_body json_schema",515 "response_format": {"type": "json_object"},516 "extra_body": {"json_schema": _PRODUCT_JSON_OBJECT_SCHEMA},517 "apply_stage": "always",518 "messages": [519 {520 "role": "system",521 "content": (522 "Extract structured data from the provided text according to the "523 "JSON schema. Return only valid JSON matching the schema exactly."524 ),525 },526 {527 "role": "user",528 "content": "Product: Wireless Headphones, ID: 101, In Stock: Yes",529 },530 ],531 "validate": lambda parsed, tcs, raw: _validate_product_json_object(parsed),532}533 534 535def _validate_product_json_object(parsed):536 if not isinstance(parsed, dict):537 return False, f"expected JSON object, got {type(parsed).__name__}: {parsed!r}"538 if not parsed:539 return False, f"expected non-empty object, got {parsed!r}"540 return True, f"product object with {len(parsed)} field(s): {sorted(parsed.keys())}"541 542 543# ---- Test 3: Nested recipe schema (always) ----544 545_RECIPE_SCHEMA = {546 "type": "json_schema",547 "json_schema": {548 "name": "recipe",549 "strict": True,550 "schema": {551 "type": "object",552 "additionalProperties": False,553 "properties": {554 "name": {"type": "string"},555 "servings": {"type": "integer"},556 "ingredients": {557 "type": "array",558 "minItems": 2,559 "items": {560 "type": "object",561 "additionalProperties": False,562 "properties": {563 "item": {"type": "string"},564 "quantity": {"type": "string"},565 },566 "required": ["item", "quantity"],567 },568 },569 "steps": {570 "type": "array",571 "minItems": 2,572 "items": {"type": "string"},573 },574 "prep_time_minutes": {"type": "integer"},575 },576 "required": ["name", "servings", "ingredients", "steps", "prep_time_minutes"],577 },578 },579}580 581RECIPE_TEST_CASE = {582 "name": "Nested recipe with arrays of objects",583 "response_format": _RECIPE_SCHEMA,584 "apply_stage": "always",585 "messages": [586 {587 "role": "user",588 "content": (589 "Give me a simple 4-serving scrambled eggs recipe as structured JSON. "590 "Include the recipe name, servings, ingredients (each with item and "591 "quantity), preparation steps, and total prep time in minutes."592 ),593 }594 ],595 "validate": lambda parsed, tcs, raw: _validate_recipe(parsed),596}597 598 599def _validate_recipe(parsed):600 required = {"name", "servings", "ingredients", "steps", "prep_time_minutes"}601 missing = required - parsed.keys()602 if missing:603 return False, f"Missing fields: {missing}"604 if not isinstance(parsed["name"], str) or not parsed["name"]:605 return False, "name must be a non-empty string"606 if not isinstance(parsed["servings"], int) or parsed["servings"] <= 0:607 return False, f"servings must be positive int: {parsed['servings']!r}"608 ings = parsed["ingredients"]609 if not isinstance(ings, list) or len(ings) < 2:610 return False, f"ingredients must be array of >=2: got {ings!r}"611 for i, ing in enumerate(ings):612 if not isinstance(ing, dict):613 return False, f"ingredient[{i}] is not an object: {ing!r}"614 ing_d = cast(dict[str, Any], ing)615 item_val = ing_d.get("item")616 qty_val = ing_d.get("quantity")617 if item_val is None or qty_val is None:618 return False, f"ingredient[{i}] missing item/quantity: {ing!r}"619 if not isinstance(item_val, str) or not isinstance(qty_val, str):620 return False, f"ingredient[{i}] fields must be strings: {ing!r}"621 steps = parsed["steps"]622 if not isinstance(steps, list) or len(steps) < 2:623 return False, f"steps must be array of >=2 strings: got {steps!r}"624 if not all(isinstance(s, str) and s for s in steps):625 return False, "all steps must be non-empty strings"626 pt = parsed["prep_time_minutes"]627 if not isinstance(pt, int) or pt <= 0:628 return False, f"prep_time_minutes must be positive int: {pt!r}"629 return True, f"recipe '{parsed['name']}' with {len(ings)} ingredients, {len(steps)} steps"630 631 632# ---- Test 4: Tool call -> structured product comparison (after_tools) ----633 634_SHOP_TOOLS = [635 {636 "type": "function",637 "function": {638 "name": "search_products",639 "description": "Search a product catalogue by keyword.",640 "parameters": {641 "type": "object",642 "properties": {643 "query": {"type": "string"},644 },645 "required": ["query"],646 },647 },648 },649 {650 "type": "function",651 "function": {652 "name": "get_product_details",653 "description": "Get detailed specs for a product by ID.",654 "parameters": {655 "type": "object",656 "properties": {657 "product_id": {"type": "string"},658 },659 "required": ["product_id"],660 },661 },662 },663]664 665_SHOP_SEARCH_RESULT = {666 "results": [667 {"product_id": "LAP-001", "title": "AeroBook 13 Pro", "price": 1399.0, "rating": 4.7},668 {"product_id": "LAP-002", "title": "QuantumSlim 14", "price": 1199.0, "rating": 4.4},669 {"product_id": "LAP-003", "title": "NimbusWork Ultra 15", "price": 999.0, "rating": 4.2},670 ],671}672_SHOP_PRODUCT_DETAILS = {673 "LAP-001": {674 "product_id": "LAP-001",675 "title": "AeroBook 13 Pro",676 "cpu": "M-series 10-core",677 "ram_gb": 16,678 "storage_gb": 512,679 "battery_hours": 18,680 "weight_kg": 1.24,681 "price": 1399.0,682 },683 "LAP-002": {684 "product_id": "LAP-002",685 "title": "QuantumSlim 14",686 "cpu": "Core i7 12-core",687 "ram_gb": 16,688 "storage_gb": 512,689 "battery_hours": 12,690 "weight_kg": 1.35,691 "price": 1199.0,692 },693 "LAP-003": {694 "product_id": "LAP-003",695 "title": "NimbusWork Ultra 15",696 "cpu": "Ryzen 7 8-core",697 "ram_gb": 16,698 "storage_gb": 1024,699 "battery_hours": 10,700 "weight_kg": 1.70,701 "price": 999.0,702 },703}704 705 706def _shop_details_mock(args):707 pid = args.get("product_id", "")708 if pid in _SHOP_PRODUCT_DETAILS:709 return json.dumps(_SHOP_PRODUCT_DETAILS[pid])710 return json.dumps({"error": f"unknown product_id: {pid}"})711 712 713_SHOP_COMPARISON_SCHEMA = {714 "type": "json_schema",715 "json_schema": {716 "name": "laptop_comparison",717 "strict": True,718 "schema": {719 "type": "object",720 "additionalProperties": False,721 "properties": {722 "recommendation": {"type": "string"},723 "ranked_candidates": {724 "type": "array",725 "minItems": 2,726 "items": {727 "type": "object",728 "additionalProperties": False,729 "properties": {730 "product_id": {"type": "string"},731 "title": {"type": "string"},732 "score": {"type": "number"},733 "reason": {"type": "string"},734 },735 "required": ["product_id", "title", "score", "reason"],736 },737 },738 },739 "required": ["recommendation", "ranked_candidates"],740 },741 },742}743 744SHOP_COMPARISON_TEST_CASE = {745 "name": "Tool calls then structured laptop comparison (after_tools)",746 "response_format": _SHOP_COMPARISON_SCHEMA,747 "apply_stage": "after_tools",748 "tools": _SHOP_TOOLS,749 "mock_tool_responses": {750 "search_products": lambda _: json.dumps(_SHOP_SEARCH_RESULT),751 "get_product_details": _shop_details_mock,752 },753 "messages": [754 {755 "role": "user",756 "content": (757 "I need a lightweight laptop for travel. Please search the catalogue "758 "for 'ultraportable laptop', then fetch detailed specs for at least two "759 "of the top candidates. Once you've gathered the data I'll ask you to "760 "produce a structured comparison."761 ),762 }763 ],764 "followup": (765 "Thanks. Now produce the final comparison strictly as JSON matching the "766 "laptop_comparison schema: your single best recommendation (the product_id), "767 "and a ranked_candidates array of at least two laptops, each with "768 "product_id, title, a numeric score, and a short reason."769 ),770 "validate": lambda parsed, tcs, raw: _validate_shop_comparison(parsed, tcs),771}772 773 774def _validate_shop_comparison(parsed, tcs):775 names = [tc["function"]["name"] for tc in tcs]776 if "search_products" not in names:777 return False, f"expected search_products tool call, got {names}"778 if "get_product_details" not in names:779 return False, f"expected get_product_details tool call, got {names}"780 if "recommendation" not in parsed or not isinstance(parsed["recommendation"], str):781 return False, f"recommendation missing or not a string: {parsed!r}"782 cands = parsed.get("ranked_candidates")783 if not isinstance(cands, list) or len(cands) < 2:784 return False, f"ranked_candidates must be >=2: {cands!r}"785 valid_ids = set(_SHOP_PRODUCT_DETAILS.keys())786 candidate_pids: list = []787 for i, c in enumerate(cands):788 if not isinstance(c, dict):789 return False, f"candidate[{i}] not an object: {c!r}"790 c_d = cast(dict[str, Any], c)791 pid = c_d.get("product_id")792 title = c_d.get("title")793 score = c_d.get("score")794 reason = c_d.get("reason")795 for k, v in (("product_id", pid), ("title", title),796 ("score", score), ("reason", reason)):797 if v is None:798 return False, f"candidate[{i}] missing {k}: {c!r}"799 if pid not in valid_ids:800 return False, f"candidate[{i}].product_id not in catalogue: {pid!r}"801 if not isinstance(score, (int, float)):802 return False, f"candidate[{i}].score not numeric: {score!r}"803 candidate_pids.append(pid)804 recommendation = parsed["recommendation"]805 if recommendation not in valid_ids and recommendation not in candidate_pids:806 return False, f"recommendation {recommendation!r} not in candidates"807 return True, (808 f"tools={names}; recommended={parsed['recommendation']}; "809 f"{len(cands)} ranked candidates"810 )811 812 813# ---- Test 5: Multi-step research then structured report (after_tools) ----814 815_RESEARCH_TOOLS = [816 {817 "type": "function",818 "function": {819 "name": "get_country_stats",820 "description": "Fetch basic statistics for a country (population, GDP, capital).",821 "parameters": {822 "type": "object",823 "properties": {824 "country": {"type": "string"},825 },826 "required": ["country"],827 },828 },829 },830 {831 "type": "function",832 "function": {833 "name": "get_climate_info",834 "description": "Fetch climate information for a country.",835 "parameters": {836 "type": "object",837 "properties": {838 "country": {"type": "string"},839 },840 "required": ["country"],841 },842 },843 },844]845 846_COUNTRY_STATS = {847 "norway": {848 "country": "Norway",849 "capital": "Oslo",850 "population": 5_480_000,851 "gdp_usd_trillion": 0.48,852 "currency": "NOK",853 }854}855_CLIMATE_INFO = {856 "norway": {857 "country": "Norway",858 "climate_zone": "subarctic / temperate coastal",859 "avg_winter_temp_c": -4.5,860 "avg_summer_temp_c": 16.0,861 "annual_precipitation_mm": 1400,862 }863}864 865 866def _country_stats_mock(args):867 c = args.get("country", "").strip().lower()868 if c in _COUNTRY_STATS:869 return json.dumps(_COUNTRY_STATS[c])870 return json.dumps({"error": f"unknown country: {c}"})871 872 873def _climate_info_mock(args):874 c = args.get("country", "").strip().lower()875 if c in _CLIMATE_INFO:876 return json.dumps(_CLIMATE_INFO[c])877 return json.dumps({"error": f"unknown country: {c}"})878 879 880_RESEARCH_REPORT_SCHEMA = {881 "type": "json_schema",882 "json_schema": {883 "name": "country_report",884 "strict": True,885 "schema": {886 "type": "object",887 "additionalProperties": False,888 "properties": {889 "country": {"type": "string"},890 "capital": {"type": "string"},891 "population": {"type": "integer"},892 "climate_summary": {"type": "string"},893 "highlights": {894 "type": "array",895 "minItems": 2,896 "maxItems": 5,897 "items": {"type": "string"},898 },899 "suitable_for_tourism": {"type": "boolean"},900 },901 "required": [902 "country", "capital", "population",903 "climate_summary", "highlights", "suitable_for_tourism",904 ],905 },906 },907}908 909COUNTRY_REPORT_TEST_CASE = {910 "name": "Research pipeline then structured country report (after_tools)",911 "response_format": _RESEARCH_REPORT_SCHEMA,912 "apply_stage": "after_tools",913 "tools": _RESEARCH_TOOLS,914 "mock_tool_responses": {915 "get_country_stats": _country_stats_mock,916 "get_climate_info": _climate_info_mock,917 },918 "messages": [919 {920 "role": "user",921 "content": (922 "I'm preparing a short briefing on Norway. Please call the "923 "get_country_stats and get_climate_info tools to gather data "924 "first. Afterwards I'll ask for a structured summary."925 ),926 }927 ],928 "followup": (929 "Based on the tool results, produce the briefing as JSON matching the "930 "country_report schema. Populate every required field and provide between "931 "two and five highlights."932 ),933 "validate": lambda parsed, tcs, raw: _validate_country_report(parsed, tcs),934}935 936 937def _validate_country_report(parsed, tcs):938 names = [tc["function"]["name"] for tc in tcs]939 for required_tool in ("get_country_stats", "get_climate_info"):940 if required_tool not in names:941 return False, f"missing tool call {required_tool!r}: got {names}"942 required = {943 "country", "capital", "population",944 "climate_summary", "highlights", "suitable_for_tourism",945 }946 missing = required - parsed.keys()947 if missing:948 return False, f"missing report fields: {missing}"949 if "norway" not in parsed["country"].lower():950 return False, f"country should reference Norway: {parsed['country']!r}"951 if "oslo" not in parsed["capital"].lower():952 return False, f"capital should be Oslo: {parsed['capital']!r}"953 if not isinstance(parsed["population"], int) or parsed["population"] < 1_000_000:954 return False, f"population implausible: {parsed['population']!r}"955 if not isinstance(parsed["climate_summary"], str) or not parsed["climate_summary"]:956 return False, "climate_summary must be a non-empty string"957 hls = parsed["highlights"]958 if not isinstance(hls, list) or not (2 <= len(hls) <= 5):959 return False, f"highlights length out of range: {hls!r}"960 if not all(isinstance(h, str) and h for h in hls):961 return False, "each highlight must be a non-empty string"962 if not isinstance(parsed["suitable_for_tourism"], bool):963 return False, f"suitable_for_tourism must be bool: {parsed['suitable_for_tourism']!r}"964 return True, (965 f"tools={names}; report for {parsed['country']} "966 f"(pop {parsed['population']}, {len(hls)} highlights)"967 )968 969 970# ---------------------------------------------------------------------------971# All test cases972# ---------------------------------------------------------------------------973 974ALL_TEST_CASES = [975 BOOK_TEST_CASE,976 SENTIMENT_TEST_CASE,977 PRODUCT_JSON_OBJECT_TEST_CASE,978 RECIPE_TEST_CASE,979 SHOP_COMPARISON_TEST_CASE,980 COUNTRY_REPORT_TEST_CASE,981]982 983 984# ---------------------------------------------------------------------------985# Entry point986# ---------------------------------------------------------------------------987 988 989def main():990 parser = argparse.ArgumentParser(991 description="Test llama-server structured-output capability."992 )993 parser.add_argument("--host", default="localhost")994 parser.add_argument("--port", default=8080, type=int)995 parser.add_argument(996 "--no-stream", action="store_true", help="Disable streaming mode tests"997 )998 parser.add_argument(999 "--stream-only", action="store_true", help="Only run streaming mode tests"1000 )1001 parser.add_argument(1002 "--test",1003 help="Run only the test whose name contains this substring (case-insensitive)",1004 )1005 args = parser.parse_args()1006 1007 url = f"http://{args.host}:{args.port}/v1/chat/completions"1008 print_info(f"Testing server at {url}")1009 1010 modes: list[bool] = []1011 if not args.stream_only:1012 modes.append(False)1013 if not args.no_stream:1014 modes.append(True)1015 1016 cases: list[dict] = ALL_TEST_CASES1017 if args.test:1018 name_filter = args.test.lower()1019 cases = [c for c in cases if name_filter in str(c["name"]).lower()]1020 if not cases:1021 print_fail(f"No test cases matched '{args.test}'")1022 sys.exit(1)1023 1024 total = 01025 passed = 01026 for stream in modes:1027 for case in cases:1028 total += 11029 if run_test(url, case, stream=stream):1030 passed += 11031 1032 color = GREEN if passed == total else RED1033 _print(f"\n{BOLD}{color}{'─' * 60}{RESET}")1034 _print(f"{BOLD}{color} Results: {passed}/{total} passed{RESET}")1035 _print(f"{BOLD}{color}{'─' * 60}{RESET}\n")1036 sys.exit(0 if passed == total else 1)1037 1038 1039if __name__ == "__main__":1040 main()1041 