Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#!/usr/bin/env python32"""3Test tool calling capability via chat completions endpoint.4 5Each test case contains:6 - tools: list of tool definitions (OpenAI-compatible)7 - messages: initial conversation messages8 - mock_tool_responses: dict mapping tool_name -> callable(arguments) -> str (JSON)9 - validate: callable(tool_calls_history, final_content) -> (passed: bool, reason: str)10"""11 12import argparse13import json14import requests15import sys16 17# ---------------------------------------------------------------------------18# Color / formatting helpers19# ---------------------------------------------------------------------------20 21RESET = "\x1b[0m"22BOLD = "\x1b[1m"23DIM = "\x1b[2m"24# Foreground colors25CYAN = "\x1b[36m"26YELLOW = "\x1b[33m"27GREEN = "\x1b[32m"28RED = "\x1b[31m"29BLUE = "\x1b[34m"30WHITE = "\x1b[97m"31 32 33def _print(text="", end="\n"):34 sys.stdout.write(text + end)35 sys.stdout.flush()36 37 38def print_header(title):39 bar = "─" * 6040 _print(f"\n{BOLD}{CYAN}┌{bar}┐{RESET}")41 _print(42 f"{BOLD}{CYAN}│ {WHITE}{title}{CYAN}{' ' * max(0, 58 - len(title))}│{RESET}"43 )44 _print(f"{BOLD}{CYAN}└{bar}┘{RESET}")45 46 47def print_tool_call(name, args):48 args_str = json.dumps(args)49 _print(50 f"\n {BOLD}{YELLOW}⚙ tool call{RESET} {CYAN}{name}{RESET}{DIM}({args_str}){RESET}"51 )52 53 54def print_tool_result(result):55 preview = result[:160] + ("…" if len(result) > 160 else "")56 _print(f" {DIM}{BLUE}↳ result{RESET} {DIM}{preview}{RESET}")57 58 59def print_model_output(text):60 # printed inline during streaming; prefix with a visual marker on first chunk61 sys.stdout.write(text)62 sys.stdout.flush()63 64 65def print_pass(reason):66 _print(f"\n{BOLD}{GREEN}✔ PASS{RESET} {reason}")67 68 69def print_fail(reason):70 _print(f"\n{BOLD}{RED}✘ FAIL{RESET} {reason}")71 72 73def print_info(msg):74 _print(f"{DIM}{msg}{RESET}")75 76 77# ---------------------------------------------------------------------------78# HTTP helpers79# ---------------------------------------------------------------------------80 81 82def chat_completion(url, messages, tools=None, stream=False, force_tools=False):83 payload = {84 "messages": messages,85 "stream": stream,86 "max_tokens": 4096,87 }88 if tools:89 payload["tools"] = tools90 if force_tools:91 payload["tool_choice"] = "required"92 else:93 payload["tool_choice"] = "auto"94 95 try:96 response = requests.post(url, json=payload, stream=stream)97 response.raise_for_status()98 except requests.exceptions.RequestException as e:99 body = e.response.content if (e.response is not None) else b""100 print_fail(f"Request error: {e} | body: {body}")101 return None102 103 full_content = ""104 reasoning_content = ""105 tool_calls: list[dict] = []106 107 if stream:108 for line in response.iter_lines():109 if not line:110 continue111 decoded = line.decode("utf-8")112 if not decoded.startswith("data: "):113 continue114 data_str = decoded[6:]115 if data_str == "[DONE]":116 break117 try:118 data = json.loads(data_str)119 except json.JSONDecodeError:120 continue121 choices = data.get("choices", [])122 if not choices:123 continue124 delta = choices[0].get("delta", {})125 if delta.get("reasoning_content"):126 reasoning_content += delta["reasoning_content"]127 if delta.get("content"):128 full_content += delta["content"]129 print_model_output(delta["content"])130 for tc in delta.get("tool_calls", []):131 idx = tc.get("index", 0)132 while len(tool_calls) <= idx:133 tool_calls.append(134 {135 "id": "",136 "type": "function",137 "function": {"name": "", "arguments": ""},138 }139 )140 if "id" in tc:141 tool_calls[idx]["id"] += tc["id"]142 if "function" in tc:143 if "name" in tc["function"]:144 tool_calls[idx]["function"]["name"] += tc["function"]["name"]145 if "arguments" in tc["function"]:146 tool_calls[idx]["function"]["arguments"] += tc["function"][147 "arguments"148 ]149 else:150 data = response.json()151 choices = data.get("choices", [])152 if choices:153 msg = choices[0].get("message", {})154 full_content = msg.get("content") or ""155 reasoning_content = msg.get("reasoning_content") or ""156 tool_calls = msg.get("tool_calls") or []157 if full_content:158 print_model_output(full_content)159 160 result = {"content": full_content, "tool_calls": tool_calls}161 if reasoning_content:162 result["reasoning_content"] = reasoning_content163 return result164 165 166def all_tools_called(tools, all_tool_calls):167 all_tool_names = set([tc["function"]["name"] for tc in tools])168 all_called_tool_names = set([tc["function"]["name"] for tc in all_tool_calls])169 return all_tool_names == all_called_tool_names170 171 172def run_agentic_loop(url, messages, tools, mock_tool_responses, stream, max_turns=6, force_tools=False):173 """174 Drive the multi-turn tool-call loop:175 1. Send messages to model.176 2. If the model returns tool calls, execute mocks and append results.177 3. Repeat until no more tool calls or max_turns reached.178 179 Returns (all_tool_calls, final_content).180 """181 msgs = list(messages)182 all_tool_calls: list[dict] = []183 184 for t in range(max_turns):185 result = chat_completion(url, msgs, tools=tools, stream=stream, force_tools=(force_tools and not all_tools_called(tools, all_tool_calls)))186 if result is None:187 return all_tool_calls, None188 189 tcs = result.get("tool_calls") or []190 content = result.get("content") or ""191 192 if not tcs:193 # Print a visual separator before the final model response194 if content:195 _print(f"\n{DIM}{'·'*60}{RESET}")196 _print(f"{DIM} model response:{RESET}\n")197 return all_tool_calls, content198 199 # Record tool calls for validation200 all_tool_calls.extend(tcs)201 202 # Append assistant message with tool calls203 assistant_msg: dict = {204 "role": "assistant",205 "content": content,206 "tool_calls": tcs,207 }208 reasoning = result.get("reasoning_content")209 if reasoning:210 assistant_msg["reasoning_content"] = reasoning211 msgs.append(assistant_msg)212 213 # Execute each tool call via mock and append tool result messages214 for tc in tcs:215 tool_name = tc["function"]["name"]216 try:217 args = json.loads(tc["function"]["arguments"])218 except json.JSONDecodeError:219 args = {}220 221 print_tool_call(tool_name, args)222 223 mock_fn = mock_tool_responses.get(tool_name)224 if mock_fn:225 tool_result = mock_fn(args)226 else:227 tool_result = json.dumps({"error": f"Unknown tool: {tool_name}"})228 229 print_tool_result(tool_result)230 231 msgs.append(232 {233 "role": "tool",234 "tool_call_id": tc.get("id", ""),235 "content": tool_result,236 }237 )238 239 return all_tool_calls, None240 241 242# ---------------------------------------------------------------------------243# Test case runner244# ---------------------------------------------------------------------------245 246 247def run_test(url, test_case, stream, force_tools):248 name = test_case["name"]249 mode = f"{'stream' if stream else 'non-stream'}"250 print_header(f"{name} [{mode}, force_tools={force_tools}] ")251 252 all_tool_calls, final_content = run_agentic_loop(253 url,254 messages=test_case["messages"],255 tools=test_case["tools"],256 mock_tool_responses=test_case["mock_tool_responses"],257 stream=stream,258 force_tools=force_tools259 )260 261 if final_content is None and not all_tool_calls:262 print_fail("No response from server.")263 return False264 265 passed, reason = test_case["validate"](all_tool_calls, final_content)266 if passed:267 print_pass(reason)268 else:269 print_fail(reason)270 return passed271 272 273# ---------------------------------------------------------------------------274# Test case definitions275# ---------------------------------------------------------------------------276 277# ---- Test 1: E-commerce multi-step search (Azzoo = anonymized marketplace) ----278 279_AZZOO_TOOLS = [280 {281 "type": "function",282 "function": {283 "name": "azzoo_search_products",284 "description": (285 "Search for products on Azzoo marketplace by keyword. "286 "Returns a list of matching products with IDs, titles, ratings and prices."287 ),288 "parameters": {289 "type": "object",290 "properties": {291 "query": {292 "type": "string",293 "description": "Search keyword or phrase",294 },295 "page": {296 "type": "string",297 "description": "Page number (1-based)",298 "default": "1",299 },300 },301 "required": ["query"],302 },303 },304 },305 {306 "type": "function",307 "function": {308 "name": "azzoo_get_product",309 "description": "Retrieve detailed information about a specific Azzoo product including specs and price.",310 "parameters": {311 "type": "object",312 "properties": {313 "product_id": {314 "type": "string",315 "description": "Azzoo product identifier (e.g. AZB12345)",316 },317 },318 "required": ["product_id"],319 },320 },321 },322 {323 "type": "function",324 "function": {325 "name": "azzoo_get_reviews",326 "description": "Fetch customer reviews for an Azzoo product.",327 "parameters": {328 "type": "object",329 "properties": {330 "product_id": {331 "type": "string",332 "description": "Azzoo product identifier",333 },334 "page": {335 "type": "string",336 "description": "Review page number",337 "default": "1",338 },339 },340 "required": ["product_id"],341 },342 },343 },344]345 346_AZZOO_SEARCH_RESULT = {347 "results": [348 {349 "product_id": "AZB00001",350 "title": "SteelBrew Pro Kettle 1.7L",351 "rating": 4.6,352 "price": 34.99,353 },354 {355 "product_id": "AZB00002",356 "title": "HeatKeep Gooseneck Kettle",357 "rating": 4.3,358 "price": 27.50,359 },360 {361 "product_id": "AZB00003",362 "title": "QuickBoil Stainless Kettle",363 "rating": 4.1,364 "price": 21.00,365 },366 ]367}368_AZZOO_PRODUCT_RESULT = {369 "product_id": "AZB00001",370 "title": "SteelBrew Pro Kettle 1.7L",371 "price": 34.99,372 "rating": 4.6,373 "review_count": 2847,374 "specs": {375 "material": "18/8 stainless steel",376 "capacity": "1.7 L",377 "auto_shutoff": True,378 "keep_warm": "30 min",379 "warranty": "2 years",380 },381}382_AZZOO_REVIEWS_RESULT = {383 "product_id": "AZB00001",384 "average_rating": 4.6,385 "reviews": [386 {387 "rating": 5,388 "title": "Excellent build quality",389 "body": "Very sturdy, boils fast and stays warm longer than expected.",390 },391 {392 "rating": 5,393 "title": "Great for loose-leaf tea",394 "body": "The wide spout makes filling a teapot easy. No leaks after months of use.",395 },396 {397 "rating": 3,398 "title": "Minor lid issue",399 "body": "The lid doesn't always click shut properly, but overall happy with it.",400 },401 {402 "rating": 4,403 "title": "Good value",404 "body": "Heats quickly and the auto shutoff works reliably.",405 },406 ],407}408 409AZZOO_TEST_CASE = {410 "name": "Azzoo E-commerce: search -> product detail -> reviews",411 "messages": [412 {413 "role": "user",414 "content": (415 "I need a durable stainless steel tea kettle for my weekly tea gatherings. "416 "Please search Azzoo for 'stainless steel tea kettle', then get full details "417 "on the top-rated result, and finally fetch its customer reviews so I can "418 "check for recurring complaints. Give me a summary with pros and cons."419 ),420 }421 ],422 "tools": _AZZOO_TOOLS,423 "mock_tool_responses": {424 "azzoo_search_products": lambda _: json.dumps(_AZZOO_SEARCH_RESULT),425 "azzoo_get_product": lambda _: json.dumps(_AZZOO_PRODUCT_RESULT),426 "azzoo_get_reviews": lambda _: json.dumps(_AZZOO_REVIEWS_RESULT),427 },428 "validate": lambda tcs, content: _validate_azzoo(tcs, content),429}430 431 432def _validate_azzoo(tcs, content):433 names = [tc["function"]["name"] for tc in tcs]434 if not names:435 return False, "No tool calls made"436 if "azzoo_search_products" not in names:437 return False, f"Expected azzoo_search_products to be called, got: {names}"438 # After search the model should look up product details439 if "azzoo_get_product" not in names and "azzoo_get_reviews" not in names:440 return False, f"Expected follow-up product/review lookup, got: {names}"441 # Verify product lookup used an ID from search results442 for tc in tcs:443 if tc["function"]["name"] == "azzoo_get_product":444 try:445 args = json.loads(tc["function"]["arguments"])446 pid = args.get("product_id", "")447 if not pid:448 return False, "azzoo_get_product called with empty product_id"449 except json.JSONDecodeError:450 return False, "azzoo_get_product arguments are not valid JSON"451 if not content:452 return False, "No final summary produced"453 return True, f"All expected tools called in order: {names}"454 455 456# ---- Test 2: Fitness BMI + exercise recommendations ----457 458_FITNESS_TOOLS = [459 {460 "type": "function",461 "function": {462 "name": "calculate_bmi",463 "description": "Calculate Body Mass Index (BMI) from weight and height.",464 "parameters": {465 "type": "object",466 "properties": {467 "weight_kg": {468 "type": "number",469 "description": "Body weight in kilograms",470 },471 "height_m": {"type": "number", "description": "Height in meters"},472 },473 "required": ["weight_kg", "height_m"],474 },475 },476 },477 {478 "type": "function",479 "function": {480 "name": "get_exercises",481 "description": (482 "Fetch a list of exercises filtered by muscle group, difficulty, category, "483 "and/or force type."484 ),485 "parameters": {486 "type": "object",487 "properties": {488 "muscle": {489 "type": "string",490 "description": "Target muscle group (e.g. chest, back, legs)",491 },492 "difficulty": {493 "type": "string",494 "description": "Difficulty level: beginner, intermediate, expert",495 },496 "category": {497 "type": "string",498 "description": "Exercise category (e.g. strength, cardio, stretching)",499 },500 "force": {501 "type": "string",502 "description": "Force type: push, pull, static",503 },504 },505 "required": [],506 },507 },508 },509]510 511_BMI_RESULT = {"bmi": 24.5, "category": "Normal weight", "healthy_range": "18.5 – 24.9"}512_EXERCISES_RESULT = {513 "exercises": [514 {515 "name": "Push-Up",516 "muscle": "chest",517 "difficulty": "beginner",518 "equipment": "none",519 "instructions": "Keep body straight, lower chest to floor.",520 },521 {522 "name": "Incline Dumbbell Press",523 "muscle": "chest",524 "difficulty": "beginner",525 "equipment": "dumbbells, bench",526 "instructions": "Press dumbbells up from chest on incline bench.",527 },528 {529 "name": "Chest Fly (cables)",530 "muscle": "chest",531 "difficulty": "beginner",532 "equipment": "cable machine",533 "instructions": "Bring cables together in an arc motion.",534 },535 ]536}537 538FITNESS_TEST_CASE = {539 "name": "Fitness: BMI calculation + exercise suggestions",540 "messages": [541 {542 "role": "user",543 "content": (544 "I'm a 32-year-old male, 78 kg and 1.80 m tall. "545 "Please calculate my BMI and then suggest some beginner chest exercises I can do "546 "to build strength. Give me a short personalised plan."547 ),548 }549 ],550 "tools": _FITNESS_TOOLS,551 "mock_tool_responses": {552 "calculate_bmi": lambda _: json.dumps(_BMI_RESULT),553 "get_exercises": lambda _: json.dumps(_EXERCISES_RESULT),554 },555 "validate": lambda tcs, content: _validate_fitness(tcs, content),556}557 558 559def _validate_fitness(tcs, content):560 names = [tc["function"]["name"] for tc in tcs]561 if not names:562 return False, "No tool calls made"563 if "calculate_bmi" not in names:564 return False, f"Expected calculate_bmi to be called, got: {names}"565 # Validate BMI args contain plausible values566 for tc in tcs:567 if tc["function"]["name"] == "calculate_bmi":568 try:569 args = json.loads(tc["function"]["arguments"])570 w = args.get("weight_kg")571 h = args.get("height_m")572 if w is None or h is None:573 return False, f"calculate_bmi missing weight_kg or height_m: {args}"574 if not (50 <= float(w) <= 200):575 return False, f"calculate_bmi weight out of plausible range: {w}"576 if not (1.0 <= float(h) <= 2.5):577 return False, f"calculate_bmi height out of plausible range: {h}"578 except (json.JSONDecodeError, ValueError) as e:579 return False, f"calculate_bmi argument error: {e}"580 if not content:581 return False, "No final plan produced"582 return True, f"Tools called: {names}"583 584 585# ---- Test 3: Community class planning (anonymised cooking/topic discovery) ----586 587_COMMUNITY_TOOLS = [588 {589 "type": "function",590 "function": {591 "name": "get_trending_questions",592 "description": (593 "Fetch commonly asked questions on a topic from search engine 'People Also Ask' boxes."594 ),595 "parameters": {596 "type": "object",597 "properties": {598 "query": {"type": "string", "description": "Topic to search for"},599 "max_results": {600 "type": "integer",601 "description": "Maximum questions to return",602 "default": 10,603 },604 },605 "required": ["query"],606 },607 },608 },609 {610 "type": "function",611 "function": {612 "name": "search_mobile_apps",613 "description": "Search the mobile app store for apps matching a category or keyword.",614 "parameters": {615 "type": "object",616 "properties": {617 "keyword": {618 "type": "string",619 "description": "Search keyword (e.g. 'Italian cooking')",620 },621 "platform": {622 "type": "string",623 "enum": ["ios", "android", "both"],624 "default": "both",625 },626 "max_results": {627 "type": "integer",628 "description": "Number of results",629 "default": 10,630 },631 },632 "required": ["keyword"],633 },634 },635 },636]637 638_TRENDING_QUESTIONS_RESULT = {639 "query": "Italian cuisine",640 "questions": [641 "What are the most popular Italian dishes?",642 "What makes Italian food different from other cuisines?",643 "How do you make authentic Italian pasta from scratch?",644 "What are traditional Italian desserts?",645 "What herbs are commonly used in Italian cooking?",646 "Is Italian food healthy?",647 "What wine pairs best with Italian pasta?",648 ],649}650_APPS_RESULT = {651 "keyword": "Italian cooking",652 "results": [653 {654 "name": "PastaPro",655 "rating": 4.5,656 "installs": "500K+",657 "focus": "pasta recipes only",658 },659 {660 "name": "CookEasy",661 "rating": 4.2,662 "installs": "1M+",663 "focus": "general cooking, limited Italian content",664 },665 {666 "name": "ItalianKitchen",667 "rating": 3.8,668 "installs": "100K+",669 "focus": "regional Italian recipes, no video",670 },671 ],672}673 674COMMUNITY_CLASS_TEST_CASE = {675 "name": "Community class planning: trending topics + app gap analysis",676 "messages": [677 {678 "role": "user",679 "content": (680 "I want to start teaching Italian cooking classes at my community centre. "681 "First, find out what people commonly ask about Italian cuisine online. "682 "Then search for existing Italian cooking apps to see what they cover. "683 "Use both results to suggest three unique angles for my classes that fill gaps "684 "in what apps already offer."685 ),686 }687 ],688 "tools": _COMMUNITY_TOOLS,689 "mock_tool_responses": {690 "get_trending_questions": lambda _: json.dumps(_TRENDING_QUESTIONS_RESULT),691 "search_mobile_apps": lambda _: json.dumps(_APPS_RESULT),692 },693 "validate": lambda tcs, content: _validate_community(tcs, content),694}695 696 697def _validate_community(tcs, content):698 names = [tc["function"]["name"] for tc in tcs]699 if not names:700 return False, "No tool calls made"701 missing = [702 t for t in ("get_trending_questions", "search_mobile_apps") if t not in names703 ]704 if missing:705 return False, f"Missing expected tool calls: {missing}; got: {names}"706 if not content:707 return False, "No class suggestion produced"708 return True, f"Both discovery tools called: {names}"709 710 711# ---- Test 4: Multi-hostname geolocation filter (anonymized gallery discovery) ----712# Inspired by: checking gallery website server locations to find truly remote venues.713# Anonymized: galleryone.de → halle-eins.de, gallerytwo.fr → galerie-deux.fr,714# gallerythree.it → galleria-tre.it715 716_GEO_TOOLS = [717 {718 "type": "function",719 "function": {720 "name": "lookup_ip_geolocation",721 "description": (722 "Retrieve geolocation data for an IP address or hostname, including country, "723 "city, coordinates, and network info. Useful for verifying physical server "724 "locations or personalising regional content."725 ),726 "parameters": {727 "type": "object",728 "properties": {729 "host": {730 "type": "string",731 "description": "IP address or hostname to look up (e.g. '8.8.8.8' or 'example.com').",732 },733 },734 "required": ["host"],735 },736 },737 },738]739 740# Mock: one urban (Berlin → discard), two rural (keep)741_GEO_RESPONSES = {742 "halle-eins.de": {743 "host": "halle-eins.de",744 "city": "Berlin",745 "country": "DE",746 "lat": 52.5200,747 "lon": 13.4050,748 "is_major_city": True,749 },750 "galerie-deux.fr": {751 "host": "galerie-deux.fr",752 "city": "Rocamadour",753 "country": "FR",754 "lat": 44.7994,755 "lon": 1.6178,756 "is_major_city": False,757 },758 "galleria-tre.it": {759 "host": "galleria-tre.it",760 "city": "Matera",761 "country": "IT",762 "lat": 40.6664,763 "lon": 16.6044,764 "is_major_city": False,765 },766}767 768 769def _geo_mock(args):770 host = args.get("host", "")771 return json.dumps(_GEO_RESPONSES.get(host, {"error": f"unknown host: {host}"}))772 773 774GEO_TEST_CASE = {775 "name": "Gallery geolocation: filter urban venues, keep remote ones",776 "messages": [777 {778 "role": "user",779 "content": (780 "I have abstract paintings to exhibit in remote European galleries. "781 "I received enquiries from three venues: halle-eins.de, galerie-deux.fr, "782 "and galleria-tre.it. Please look up the geolocation of each website's server. "783 "Discard any venue whose server is in a major city (e.g. Berlin, Paris, Rome). "784 "For the remaining venues, report their exact coordinates so I can check "785 "whether hiking trails are nearby — my work thrives where nature and art meet."786 ),787 }788 ],789 "tools": _GEO_TOOLS,790 "mock_tool_responses": {791 "lookup_ip_geolocation": _geo_mock,792 },793 "validate": lambda tcs, content: _validate_geo(tcs, content),794}795 796 797def _validate_geo(tcs, content):798 names = [tc["function"]["name"] for tc in tcs]799 if not names:800 return False, "No tool calls made"801 # Expect exactly one geolocation call per domain (3 total)802 geo_calls = [tc for tc in tcs if tc["function"]["name"] == "lookup_ip_geolocation"]803 if len(geo_calls) < 3:804 return (805 False,806 f"Expected geolocation called 3 times (once per domain), got {len(geo_calls)}",807 )808 queried_hosts = set()809 for tc in geo_calls:810 try:811 args = json.loads(tc["function"]["arguments"])812 host = args.get("host", "")813 if not host:814 return False, f"lookup_ip_geolocation called with empty host: {args}"815 queried_hosts.add(host)816 except json.JSONDecodeError:817 return False, "lookup_ip_geolocation arguments are not valid JSON"818 expected = {"halle-eins.de", "galerie-deux.fr", "galleria-tre.it"}819 if not expected.issubset(queried_hosts):820 return (821 False,822 f"Not all domains queried. Expected {expected}, got {queried_hosts}",823 )824 if not content:825 return False, "No final summary produced"826 return True, f"All 3 domains geolocated: {sorted(queried_hosts)}"827 828 829# ---- Test 5: EV fleet expansion — stock → security → property → video ----830# Inspired by: multi-step business analysis combining finance, cybersecurity,831# real estate and educational content.832# Anonymized: Tesla → Voltara (VLTR), Rivian → Rivex (RVXN),833# Trenton → Halverton834 835_EV_TOOLS = [836 {837 "type": "function",838 "function": {839 "name": "get_stock_quote",840 "description": "Retrieve the latest market quote for a financial instrument by ticker symbol.",841 "parameters": {842 "type": "object",843 "properties": {844 "symbol": {845 "type": "string",846 "description": "Ticker symbol (e.g. 'VLTR', 'RVXN')",847 },848 "interval": {849 "type": "string",850 "description": "Time interval: 1min, 5min, 1h, 1day, 1week",851 "default": "1day",852 },853 },854 "required": ["symbol"],855 },856 },857 },858 {859 "type": "function",860 "function": {861 "name": "get_security_advisories",862 "description": (863 "Fetch current cybersecurity advisories from the national security agency, "864 "covering known vulnerabilities and exploits for industrial and consumer systems."865 ),866 "parameters": {867 "type": "object",868 "properties": {869 "keyword": {870 "type": "string",871 "description": "Filter advisories by keyword or product name",872 },873 "limit": {874 "type": "integer",875 "description": "Maximum number of advisories to return",876 "default": 5,877 },878 },879 "required": [],880 },881 },882 },883 {884 "type": "function",885 "function": {886 "name": "search_commercial_properties",887 "description": "Search for commercial properties (offices, garages, warehouses) available for rent or sale in a given city.",888 "parameters": {889 "type": "object",890 "properties": {891 "city": {"type": "string", "description": "City name to search in"},892 "property_type": {893 "type": "string",894 "description": "Type of property: office, garage, warehouse, premises",895 },896 "operation": {897 "type": "string",898 "enum": ["rent", "sale"],899 "default": "rent",900 },901 "max_price": {902 "type": "integer",903 "description": "Maximum monthly rent or sale price",904 },905 },906 "required": ["city", "property_type"],907 },908 },909 },910 {911 "type": "function",912 "function": {913 "name": "get_video_recommendations",914 "description": "Fetch a list of recommended videos related to a given topic or reference video.",915 "parameters": {916 "type": "object",917 "properties": {918 "topic": {919 "type": "string",920 "description": "Topic or keyword to search for related videos",921 },922 },923 "required": ["topic"],924 },925 },926 },927]928 929_STOCK_RESULT_VLTR = {930 "symbol": "VLTR",931 "company": "Voltara Inc.",932 "price": 218.45,933 "change_pct": "+2.3%",934 "market_cap": "694B",935 "currency": "USD",936}937_STOCK_RESULT_RVXN = {938 "symbol": "RVXN",939 "company": "Rivex Motors",940 "price": 12.80,941 "change_pct": "-1.1%",942 "market_cap": "11B",943 "currency": "USD",944}945_ADVISORIES_RESULT = {946 "count": 2,947 "advisories": [948 {949 "id": "ICSA-24-102-01",950 "title": "Voltara In-Vehicle Infotainment System Authentication Bypass",951 "severity": "Medium",952 "summary": "Improper authentication in the OTA update module may allow an adjacent attacker to install unsigned firmware.",953 "published": "2024-04-11",954 },955 {956 "id": "ICSA-24-085-03",957 "title": "Voltara Charging Management API Input Validation Flaw",958 "severity": "Low",959 "summary": "Insufficient input validation in the charging session API could expose internal error messages.",960 "published": "2024-03-26",961 },962 ],963}964_PROPERTIES_RESULT = {965 "city": "Halverton",966 "listings": [967 {968 "id": "HV-0041",969 "type": "garage",970 "area_sqm": 420,971 "monthly_rent": 2800,972 "ev_power_outlets": 12,973 "address": "14 Ironworks Lane, Halverton",974 },975 {976 "id": "HV-0089",977 "type": "warehouse",978 "area_sqm": 900,979 "monthly_rent": 4200,980 "ev_power_outlets": 30,981 "address": "7 Depot Road, Halverton",982 },983 ],984}985_VIDEOS_RESULT = {986 "topic": "fleet electrification",987 "recommendations": [988 {989 "title": "How to Build an EV Fleet from Scratch",990 "channel": "Fleet Future",991 "views": "182K",992 },993 {994 "title": "EV Charging Infrastructure for Commercial Fleets",995 "channel": "GreenDrive Pro",996 "views": "94K",997 },998 {999 "title": "Total Cost of Ownership: Electric vs Diesel Vans",1000 "channel": "LogisticsTech",1001 "views": "61K",1002 },1003 ],1004}1005 1006 1007def _ev_stock_mock(args):1008 symbol = args.get("symbol", "").upper()1009 if symbol == "VLTR":1010 return json.dumps(_STOCK_RESULT_VLTR)1011 if symbol == "RVXN":1012 return json.dumps(_STOCK_RESULT_RVXN)1013 return json.dumps({"error": f"Unknown symbol: {symbol}"})1014 1015 1016EV_FLEET_TEST_CASE = {1017 "name": "EV fleet expansion: stock → cybersecurity → property → videos",1018 "messages": [1019 {1020 "role": "user",1021 "content": (1022 "I'm expanding my courier business into electric vehicles and need a multi-step analysis:\n"1023 "1. Get the latest stock quote for Voltara (VLTR) and Rivex (RVXN). "1024 "If either is above $50, continue with that company.\n"1025 "2. Search for cybersecurity advisories related to that company's vehicle models "1026 "to understand any tech risks.\n"1027 "3. Find commercial garage or warehouse properties in Halverton suitable for "1028 "EV charging infrastructure.\n"1029 "4. Recommend videos on fleet electrification strategies.\n"1030 "Please work through all four steps and give me a concise summary."1031 ),1032 }1033 ],1034 "tools": _EV_TOOLS,1035 "mock_tool_responses": {1036 "get_stock_quote": _ev_stock_mock,1037 "get_security_advisories": lambda _: json.dumps(_ADVISORIES_RESULT),1038 "search_commercial_properties": lambda _: json.dumps(_PROPERTIES_RESULT),1039 "get_video_recommendations": lambda _: json.dumps(_VIDEOS_RESULT),1040 },1041 "validate": lambda tcs, content: _validate_ev(tcs, content),1042}1043 1044 1045def _validate_ev(tcs, content):1046 names = [tc["function"]["name"] for tc in tcs]1047 if not names:1048 return False, "No tool calls made"1049 # Stock quote must come first1050 if names[0] != "get_stock_quote":1051 return False, f"Expected get_stock_quote to be called first, got: {names[0]}"1052 stock_calls = [tc for tc in tcs if tc["function"]["name"] == "get_stock_quote"]1053 for tc in stock_calls:1054 try:1055 args = json.loads(tc["function"]["arguments"])1056 sym = args.get("symbol", "")1057 if not sym:1058 return False, f"get_stock_quote called with empty symbol: {args}"1059 except json.JSONDecodeError:1060 return False, "get_stock_quote arguments are not valid JSON"1061 # All four pipeline tools expected1062 required = [1063 "get_stock_quote",1064 "get_security_advisories",1065 "search_commercial_properties",1066 "get_video_recommendations",1067 ]1068 missing = [t for t in required if t not in names]1069 if missing:1070 return False, f"Missing pipeline steps: {missing}"1071 if not content:1072 return False, "No final summary produced"1073 return True, f"Full 4-step pipeline executed: {names}"1074 1075 1076# ---------------------------------------------------------------------------1077# All test cases1078# ---------------------------------------------------------------------------1079 1080ALL_TEST_CASES = [1081 AZZOO_TEST_CASE,1082 FITNESS_TEST_CASE,1083 COMMUNITY_CLASS_TEST_CASE,1084 GEO_TEST_CASE,1085 EV_FLEET_TEST_CASE,1086]1087 1088 1089# ---------------------------------------------------------------------------1090# Entry point1091# ---------------------------------------------------------------------------1092 1093 1094def main():1095 parser = argparse.ArgumentParser(1096 description="Test llama-server tool-calling capability."1097 )1098 parser.add_argument("--host", default="localhost")1099 parser.add_argument("--port", default=8080, type=int)1100 parser.add_argument(1101 "--no-stream", action="store_true", help="Disable streaming mode tests"1102 )1103 parser.add_argument(1104 "--stream-only", action="store_true", help="Only run streaming mode tests"1105 )1106 parser.add_argument(1107 "--force-tools", action="store_true", help="Change tool mode to forced instead of auto"1108 )1109 parser.add_argument(1110 "--test",1111 help="Run only the test whose name contains this substring (case-insensitive)",1112 )1113 args = parser.parse_args()1114 1115 url = f"http://{args.host}:{args.port}/v1/chat/completions"1116 print_info(f"Testing server at {url}")1117 1118 modes = []1119 force_tools = False1120 if not args.stream_only:1121 modes.append(False)1122 if not args.no_stream:1123 modes.append(True)1124 if args.force_tools:1125 force_tools = True1126 1127 cases: list[dict] = ALL_TEST_CASES1128 if args.test:1129 name_filter = args.test.lower()1130 cases = [c for c in cases if name_filter in str(c["name"]).lower()]1131 if not cases:1132 print_fail(f"No test cases matched '{args.test}'")1133 sys.exit(1)1134 1135 total = 01136 passed = 01137 for stream in modes:1138 for case in cases:1139 total += 11140 if run_test(url, case, stream=stream, force_tools=force_tools):1141 passed += 11142 1143 color = GREEN if passed == total else RED1144 _print(f"\n{BOLD}{color}{'─'*60}{RESET}")1145 _print(f"{BOLD}{color} Results: {passed}/{total} passed{RESET}")1146 _print(f"{BOLD}{color}{'─'*60}{RESET}\n")1147 sys.exit(0 if passed == total else 1)1148 1149 1150if __name__ == "__main__":1151 main()1152 