Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
server-test-function-call.py1152 linesDownload Raw Back to scripts
1#!/usr/bin/env python32"""3Test tool calling capability via chat completions endpoint.4 5Each test case contains:6  - tools: list of tool definitions (OpenAI-compatible)7  - messages: initial conversation messages8  - mock_tool_responses: dict mapping tool_name -> callable(arguments) -> str (JSON)9  - validate: callable(tool_calls_history, final_content) -> (passed: bool, reason: str)10"""11 12import argparse13import json14import requests15import sys16 17# ---------------------------------------------------------------------------18# Color / formatting helpers19# ---------------------------------------------------------------------------20 21RESET = "\x1b[0m"22BOLD = "\x1b[1m"23DIM = "\x1b[2m"24# Foreground colors25CYAN = "\x1b[36m"26YELLOW = "\x1b[33m"27GREEN = "\x1b[32m"28RED = "\x1b[31m"29BLUE = "\x1b[34m"30WHITE = "\x1b[97m"31 32 33def _print(text="", end="\n"):34    sys.stdout.write(text + end)35    sys.stdout.flush()36 37 38def print_header(title):39    bar = "─" * 6040    _print(f"\n{BOLD}{CYAN}┌{bar}┐{RESET}")41    _print(42        f"{BOLD}{CYAN}│  {WHITE}{title}{CYAN}{' ' * max(0, 58 - len(title))}│{RESET}"43    )44    _print(f"{BOLD}{CYAN}└{bar}┘{RESET}")45 46 47def print_tool_call(name, args):48    args_str = json.dumps(args)49    _print(50        f"\n  {BOLD}{YELLOW}⚙ tool call{RESET}  {CYAN}{name}{RESET}{DIM}({args_str}){RESET}"51    )52 53 54def print_tool_result(result):55    preview = result[:160] + ("…" if len(result) > 160 else "")56    _print(f"  {DIM}{BLUE}↳ result{RESET}    {DIM}{preview}{RESET}")57 58 59def print_model_output(text):60    # printed inline during streaming; prefix with a visual marker on first chunk61    sys.stdout.write(text)62    sys.stdout.flush()63 64 65def print_pass(reason):66    _print(f"\n{BOLD}{GREEN}✔ PASS{RESET}  {reason}")67 68 69def print_fail(reason):70    _print(f"\n{BOLD}{RED}✘ FAIL{RESET}  {reason}")71 72 73def print_info(msg):74    _print(f"{DIM}{msg}{RESET}")75 76 77# ---------------------------------------------------------------------------78# HTTP helpers79# ---------------------------------------------------------------------------80 81 82def chat_completion(url, messages, tools=None, stream=False, force_tools=False):83    payload = {84        "messages": messages,85        "stream": stream,86        "max_tokens": 4096,87    }88    if tools:89        payload["tools"] = tools90        if force_tools:91            payload["tool_choice"] = "required"92        else:93            payload["tool_choice"] = "auto"94 95    try:96        response = requests.post(url, json=payload, stream=stream)97        response.raise_for_status()98    except requests.exceptions.RequestException as e:99        body = e.response.content if (e.response is not None) else b""100        print_fail(f"Request error: {e} | body: {body}")101        return None102 103    full_content = ""104    reasoning_content = ""105    tool_calls: list[dict] = []106 107    if stream:108        for line in response.iter_lines():109            if not line:110                continue111            decoded = line.decode("utf-8")112            if not decoded.startswith("data: "):113                continue114            data_str = decoded[6:]115            if data_str == "[DONE]":116                break117            try:118                data = json.loads(data_str)119            except json.JSONDecodeError:120                continue121            choices = data.get("choices", [])122            if not choices:123                continue124            delta = choices[0].get("delta", {})125            if delta.get("reasoning_content"):126                reasoning_content += delta["reasoning_content"]127            if delta.get("content"):128                full_content += delta["content"]129                print_model_output(delta["content"])130            for tc in delta.get("tool_calls", []):131                idx = tc.get("index", 0)132                while len(tool_calls) <= idx:133                    tool_calls.append(134                        {135                            "id": "",136                            "type": "function",137                            "function": {"name": "", "arguments": ""},138                        }139                    )140                if "id" in tc:141                    tool_calls[idx]["id"] += tc["id"]142                if "function" in tc:143                    if "name" in tc["function"]:144                        tool_calls[idx]["function"]["name"] += tc["function"]["name"]145                    if "arguments" in tc["function"]:146                        tool_calls[idx]["function"]["arguments"] += tc["function"][147                            "arguments"148                        ]149    else:150        data = response.json()151        choices = data.get("choices", [])152        if choices:153            msg = choices[0].get("message", {})154            full_content = msg.get("content") or ""155            reasoning_content = msg.get("reasoning_content") or ""156            tool_calls = msg.get("tool_calls") or []157            if full_content:158                print_model_output(full_content)159 160    result = {"content": full_content, "tool_calls": tool_calls}161    if reasoning_content:162        result["reasoning_content"] = reasoning_content163    return result164 165 166def all_tools_called(tools, all_tool_calls):167    all_tool_names = set([tc["function"]["name"] for tc in tools])168    all_called_tool_names = set([tc["function"]["name"] for tc in all_tool_calls])169    return all_tool_names == all_called_tool_names170 171 172def run_agentic_loop(url, messages, tools, mock_tool_responses, stream, max_turns=6, force_tools=False):173    """174    Drive the multi-turn tool-call loop:175      1. Send messages to model.176      2. If the model returns tool calls, execute mocks and append results.177      3. Repeat until no more tool calls or max_turns reached.178 179    Returns (all_tool_calls, final_content).180    """181    msgs = list(messages)182    all_tool_calls: list[dict] = []183 184    for t in range(max_turns):185        result = chat_completion(url, msgs, tools=tools, stream=stream, force_tools=(force_tools and not all_tools_called(tools, all_tool_calls)))186        if result is None:187            return all_tool_calls, None188 189        tcs = result.get("tool_calls") or []190        content = result.get("content") or ""191 192        if not tcs:193            # Print a visual separator before the final model response194            if content:195                _print(f"\n{DIM}{'·'*60}{RESET}")196                _print(f"{DIM}  model response:{RESET}\n")197            return all_tool_calls, content198 199        # Record tool calls for validation200        all_tool_calls.extend(tcs)201 202        # Append assistant message with tool calls203        assistant_msg: dict = {204            "role": "assistant",205            "content": content,206            "tool_calls": tcs,207        }208        reasoning = result.get("reasoning_content")209        if reasoning:210            assistant_msg["reasoning_content"] = reasoning211        msgs.append(assistant_msg)212 213        # Execute each tool call via mock and append tool result messages214        for tc in tcs:215            tool_name = tc["function"]["name"]216            try:217                args = json.loads(tc["function"]["arguments"])218            except json.JSONDecodeError:219                args = {}220 221            print_tool_call(tool_name, args)222 223            mock_fn = mock_tool_responses.get(tool_name)224            if mock_fn:225                tool_result = mock_fn(args)226            else:227                tool_result = json.dumps({"error": f"Unknown tool: {tool_name}"})228 229            print_tool_result(tool_result)230 231            msgs.append(232                {233                    "role": "tool",234                    "tool_call_id": tc.get("id", ""),235                    "content": tool_result,236                }237            )238 239    return all_tool_calls, None240 241 242# ---------------------------------------------------------------------------243# Test case runner244# ---------------------------------------------------------------------------245 246 247def run_test(url, test_case, stream, force_tools):248    name = test_case["name"]249    mode = f"{'stream' if stream else 'non-stream'}"250    print_header(f"{name} [{mode}, force_tools={force_tools}] ")251 252    all_tool_calls, final_content = run_agentic_loop(253        url,254        messages=test_case["messages"],255        tools=test_case["tools"],256        mock_tool_responses=test_case["mock_tool_responses"],257        stream=stream,258        force_tools=force_tools259    )260 261    if final_content is None and not all_tool_calls:262        print_fail("No response from server.")263        return False264 265    passed, reason = test_case["validate"](all_tool_calls, final_content)266    if passed:267        print_pass(reason)268    else:269        print_fail(reason)270    return passed271 272 273# ---------------------------------------------------------------------------274# Test case definitions275# ---------------------------------------------------------------------------276 277# ---- Test 1: E-commerce multi-step search (Azzoo = anonymized marketplace) ----278 279_AZZOO_TOOLS = [280    {281        "type": "function",282        "function": {283            "name": "azzoo_search_products",284            "description": (285                "Search for products on Azzoo marketplace by keyword. "286                "Returns a list of matching products with IDs, titles, ratings and prices."287            ),288            "parameters": {289                "type": "object",290                "properties": {291                    "query": {292                        "type": "string",293                        "description": "Search keyword or phrase",294                    },295                    "page": {296                        "type": "string",297                        "description": "Page number (1-based)",298                        "default": "1",299                    },300                },301                "required": ["query"],302            },303        },304    },305    {306        "type": "function",307        "function": {308            "name": "azzoo_get_product",309            "description": "Retrieve detailed information about a specific Azzoo product including specs and price.",310            "parameters": {311                "type": "object",312                "properties": {313                    "product_id": {314                        "type": "string",315                        "description": "Azzoo product identifier (e.g. AZB12345)",316                    },317                },318                "required": ["product_id"],319            },320        },321    },322    {323        "type": "function",324        "function": {325            "name": "azzoo_get_reviews",326            "description": "Fetch customer reviews for an Azzoo product.",327            "parameters": {328                "type": "object",329                "properties": {330                    "product_id": {331                        "type": "string",332                        "description": "Azzoo product identifier",333                    },334                    "page": {335                        "type": "string",336                        "description": "Review page number",337                        "default": "1",338                    },339                },340                "required": ["product_id"],341            },342        },343    },344]345 346_AZZOO_SEARCH_RESULT = {347    "results": [348        {349            "product_id": "AZB00001",350            "title": "SteelBrew Pro Kettle 1.7L",351            "rating": 4.6,352            "price": 34.99,353        },354        {355            "product_id": "AZB00002",356            "title": "HeatKeep Gooseneck Kettle",357            "rating": 4.3,358            "price": 27.50,359        },360        {361            "product_id": "AZB00003",362            "title": "QuickBoil Stainless Kettle",363            "rating": 4.1,364            "price": 21.00,365        },366    ]367}368_AZZOO_PRODUCT_RESULT = {369    "product_id": "AZB00001",370    "title": "SteelBrew Pro Kettle 1.7L",371    "price": 34.99,372    "rating": 4.6,373    "review_count": 2847,374    "specs": {375        "material": "18/8 stainless steel",376        "capacity": "1.7 L",377        "auto_shutoff": True,378        "keep_warm": "30 min",379        "warranty": "2 years",380    },381}382_AZZOO_REVIEWS_RESULT = {383    "product_id": "AZB00001",384    "average_rating": 4.6,385    "reviews": [386        {387            "rating": 5,388            "title": "Excellent build quality",389            "body": "Very sturdy, boils fast and stays warm longer than expected.",390        },391        {392            "rating": 5,393            "title": "Great for loose-leaf tea",394            "body": "The wide spout makes filling a teapot easy. No leaks after months of use.",395        },396        {397            "rating": 3,398            "title": "Minor lid issue",399            "body": "The lid doesn't always click shut properly, but overall happy with it.",400        },401        {402            "rating": 4,403            "title": "Good value",404            "body": "Heats quickly and the auto shutoff works reliably.",405        },406    ],407}408 409AZZOO_TEST_CASE = {410    "name": "Azzoo E-commerce: search -> product detail -> reviews",411    "messages": [412        {413            "role": "user",414            "content": (415                "I need a durable stainless steel tea kettle for my weekly tea gatherings. "416                "Please search Azzoo for 'stainless steel tea kettle', then get full details "417                "on the top-rated result, and finally fetch its customer reviews so I can "418                "check for recurring complaints. Give me a summary with pros and cons."419            ),420        }421    ],422    "tools": _AZZOO_TOOLS,423    "mock_tool_responses": {424        "azzoo_search_products": lambda _: json.dumps(_AZZOO_SEARCH_RESULT),425        "azzoo_get_product": lambda _: json.dumps(_AZZOO_PRODUCT_RESULT),426        "azzoo_get_reviews": lambda _: json.dumps(_AZZOO_REVIEWS_RESULT),427    },428    "validate": lambda tcs, content: _validate_azzoo(tcs, content),429}430 431 432def _validate_azzoo(tcs, content):433    names = [tc["function"]["name"] for tc in tcs]434    if not names:435        return False, "No tool calls made"436    if "azzoo_search_products" not in names:437        return False, f"Expected azzoo_search_products to be called, got: {names}"438    # After search the model should look up product details439    if "azzoo_get_product" not in names and "azzoo_get_reviews" not in names:440        return False, f"Expected follow-up product/review lookup, got: {names}"441    # Verify product lookup used an ID from search results442    for tc in tcs:443        if tc["function"]["name"] == "azzoo_get_product":444            try:445                args = json.loads(tc["function"]["arguments"])446                pid = args.get("product_id", "")447                if not pid:448                    return False, "azzoo_get_product called with empty product_id"449            except json.JSONDecodeError:450                return False, "azzoo_get_product arguments are not valid JSON"451    if not content:452        return False, "No final summary produced"453    return True, f"All expected tools called in order: {names}"454 455 456# ---- Test 2: Fitness BMI + exercise recommendations ----457 458_FITNESS_TOOLS = [459    {460        "type": "function",461        "function": {462            "name": "calculate_bmi",463            "description": "Calculate Body Mass Index (BMI) from weight and height.",464            "parameters": {465                "type": "object",466                "properties": {467                    "weight_kg": {468                        "type": "number",469                        "description": "Body weight in kilograms",470                    },471                    "height_m": {"type": "number", "description": "Height in meters"},472                },473                "required": ["weight_kg", "height_m"],474            },475        },476    },477    {478        "type": "function",479        "function": {480            "name": "get_exercises",481            "description": (482                "Fetch a list of exercises filtered by muscle group, difficulty, category, "483                "and/or force type."484            ),485            "parameters": {486                "type": "object",487                "properties": {488                    "muscle": {489                        "type": "string",490                        "description": "Target muscle group (e.g. chest, back, legs)",491                    },492                    "difficulty": {493                        "type": "string",494                        "description": "Difficulty level: beginner, intermediate, expert",495                    },496                    "category": {497                        "type": "string",498                        "description": "Exercise category (e.g. strength, cardio, stretching)",499                    },500                    "force": {501                        "type": "string",502                        "description": "Force type: push, pull, static",503                    },504                },505                "required": [],506            },507        },508    },509]510 511_BMI_RESULT = {"bmi": 24.5, "category": "Normal weight", "healthy_range": "18.5 – 24.9"}512_EXERCISES_RESULT = {513    "exercises": [514        {515            "name": "Push-Up",516            "muscle": "chest",517            "difficulty": "beginner",518            "equipment": "none",519            "instructions": "Keep body straight, lower chest to floor.",520        },521        {522            "name": "Incline Dumbbell Press",523            "muscle": "chest",524            "difficulty": "beginner",525            "equipment": "dumbbells, bench",526            "instructions": "Press dumbbells up from chest on incline bench.",527        },528        {529            "name": "Chest Fly (cables)",530            "muscle": "chest",531            "difficulty": "beginner",532            "equipment": "cable machine",533            "instructions": "Bring cables together in an arc motion.",534        },535    ]536}537 538FITNESS_TEST_CASE = {539    "name": "Fitness: BMI calculation + exercise suggestions",540    "messages": [541        {542            "role": "user",543            "content": (544                "I'm a 32-year-old male, 78 kg and 1.80 m tall. "545                "Please calculate my BMI and then suggest some beginner chest exercises I can do "546                "to build strength. Give me a short personalised plan."547            ),548        }549    ],550    "tools": _FITNESS_TOOLS,551    "mock_tool_responses": {552        "calculate_bmi": lambda _: json.dumps(_BMI_RESULT),553        "get_exercises": lambda _: json.dumps(_EXERCISES_RESULT),554    },555    "validate": lambda tcs, content: _validate_fitness(tcs, content),556}557 558 559def _validate_fitness(tcs, content):560    names = [tc["function"]["name"] for tc in tcs]561    if not names:562        return False, "No tool calls made"563    if "calculate_bmi" not in names:564        return False, f"Expected calculate_bmi to be called, got: {names}"565    # Validate BMI args contain plausible values566    for tc in tcs:567        if tc["function"]["name"] == "calculate_bmi":568            try:569                args = json.loads(tc["function"]["arguments"])570                w = args.get("weight_kg")571                h = args.get("height_m")572                if w is None or h is None:573                    return False, f"calculate_bmi missing weight_kg or height_m: {args}"574                if not (50 <= float(w) <= 200):575                    return False, f"calculate_bmi weight out of plausible range: {w}"576                if not (1.0 <= float(h) <= 2.5):577                    return False, f"calculate_bmi height out of plausible range: {h}"578            except (json.JSONDecodeError, ValueError) as e:579                return False, f"calculate_bmi argument error: {e}"580    if not content:581        return False, "No final plan produced"582    return True, f"Tools called: {names}"583 584 585# ---- Test 3: Community class planning (anonymised cooking/topic discovery) ----586 587_COMMUNITY_TOOLS = [588    {589        "type": "function",590        "function": {591            "name": "get_trending_questions",592            "description": (593                "Fetch commonly asked questions on a topic from search engine 'People Also Ask' boxes."594            ),595            "parameters": {596                "type": "object",597                "properties": {598                    "query": {"type": "string", "description": "Topic to search for"},599                    "max_results": {600                        "type": "integer",601                        "description": "Maximum questions to return",602                        "default": 10,603                    },604                },605                "required": ["query"],606            },607        },608    },609    {610        "type": "function",611        "function": {612            "name": "search_mobile_apps",613            "description": "Search the mobile app store for apps matching a category or keyword.",614            "parameters": {615                "type": "object",616                "properties": {617                    "keyword": {618                        "type": "string",619                        "description": "Search keyword (e.g. 'Italian cooking')",620                    },621                    "platform": {622                        "type": "string",623                        "enum": ["ios", "android", "both"],624                        "default": "both",625                    },626                    "max_results": {627                        "type": "integer",628                        "description": "Number of results",629                        "default": 10,630                    },631                },632                "required": ["keyword"],633            },634        },635    },636]637 638_TRENDING_QUESTIONS_RESULT = {639    "query": "Italian cuisine",640    "questions": [641        "What are the most popular Italian dishes?",642        "What makes Italian food different from other cuisines?",643        "How do you make authentic Italian pasta from scratch?",644        "What are traditional Italian desserts?",645        "What herbs are commonly used in Italian cooking?",646        "Is Italian food healthy?",647        "What wine pairs best with Italian pasta?",648    ],649}650_APPS_RESULT = {651    "keyword": "Italian cooking",652    "results": [653        {654            "name": "PastaPro",655            "rating": 4.5,656            "installs": "500K+",657            "focus": "pasta recipes only",658        },659        {660            "name": "CookEasy",661            "rating": 4.2,662            "installs": "1M+",663            "focus": "general cooking, limited Italian content",664        },665        {666            "name": "ItalianKitchen",667            "rating": 3.8,668            "installs": "100K+",669            "focus": "regional Italian recipes, no video",670        },671    ],672}673 674COMMUNITY_CLASS_TEST_CASE = {675    "name": "Community class planning: trending topics + app gap analysis",676    "messages": [677        {678            "role": "user",679            "content": (680                "I want to start teaching Italian cooking classes at my community centre. "681                "First, find out what people commonly ask about Italian cuisine online. "682                "Then search for existing Italian cooking apps to see what they cover. "683                "Use both results to suggest three unique angles for my classes that fill gaps "684                "in what apps already offer."685            ),686        }687    ],688    "tools": _COMMUNITY_TOOLS,689    "mock_tool_responses": {690        "get_trending_questions": lambda _: json.dumps(_TRENDING_QUESTIONS_RESULT),691        "search_mobile_apps": lambda _: json.dumps(_APPS_RESULT),692    },693    "validate": lambda tcs, content: _validate_community(tcs, content),694}695 696 697def _validate_community(tcs, content):698    names = [tc["function"]["name"] for tc in tcs]699    if not names:700        return False, "No tool calls made"701    missing = [702        t for t in ("get_trending_questions", "search_mobile_apps") if t not in names703    ]704    if missing:705        return False, f"Missing expected tool calls: {missing}; got: {names}"706    if not content:707        return False, "No class suggestion produced"708    return True, f"Both discovery tools called: {names}"709 710 711# ---- Test 4: Multi-hostname geolocation filter (anonymized gallery discovery) ----712# Inspired by: checking gallery website server locations to find truly remote venues.713# Anonymized: galleryone.de → halle-eins.de, gallerytwo.fr → galerie-deux.fr,714#             gallerythree.it → galleria-tre.it715 716_GEO_TOOLS = [717    {718        "type": "function",719        "function": {720            "name": "lookup_ip_geolocation",721            "description": (722                "Retrieve geolocation data for an IP address or hostname, including country, "723                "city, coordinates, and network info. Useful for verifying physical server "724                "locations or personalising regional content."725            ),726            "parameters": {727                "type": "object",728                "properties": {729                    "host": {730                        "type": "string",731                        "description": "IP address or hostname to look up (e.g. '8.8.8.8' or 'example.com').",732                    },733                },734                "required": ["host"],735            },736        },737    },738]739 740# Mock: one urban (Berlin → discard), two rural (keep)741_GEO_RESPONSES = {742    "halle-eins.de": {743        "host": "halle-eins.de",744        "city": "Berlin",745        "country": "DE",746        "lat": 52.5200,747        "lon": 13.4050,748        "is_major_city": True,749    },750    "galerie-deux.fr": {751        "host": "galerie-deux.fr",752        "city": "Rocamadour",753        "country": "FR",754        "lat": 44.7994,755        "lon": 1.6178,756        "is_major_city": False,757    },758    "galleria-tre.it": {759        "host": "galleria-tre.it",760        "city": "Matera",761        "country": "IT",762        "lat": 40.6664,763        "lon": 16.6044,764        "is_major_city": False,765    },766}767 768 769def _geo_mock(args):770    host = args.get("host", "")771    return json.dumps(_GEO_RESPONSES.get(host, {"error": f"unknown host: {host}"}))772 773 774GEO_TEST_CASE = {775    "name": "Gallery geolocation: filter urban venues, keep remote ones",776    "messages": [777        {778            "role": "user",779            "content": (780                "I have abstract paintings to exhibit in remote European galleries. "781                "I received enquiries from three venues: halle-eins.de, galerie-deux.fr, "782                "and galleria-tre.it. Please look up the geolocation of each website's server. "783                "Discard any venue whose server is in a major city (e.g. Berlin, Paris, Rome). "784                "For the remaining venues, report their exact coordinates so I can check "785                "whether hiking trails are nearby — my work thrives where nature and art meet."786            ),787        }788    ],789    "tools": _GEO_TOOLS,790    "mock_tool_responses": {791        "lookup_ip_geolocation": _geo_mock,792    },793    "validate": lambda tcs, content: _validate_geo(tcs, content),794}795 796 797def _validate_geo(tcs, content):798    names = [tc["function"]["name"] for tc in tcs]799    if not names:800        return False, "No tool calls made"801    # Expect exactly one geolocation call per domain (3 total)802    geo_calls = [tc for tc in tcs if tc["function"]["name"] == "lookup_ip_geolocation"]803    if len(geo_calls) < 3:804        return (805            False,806            f"Expected geolocation called 3 times (once per domain), got {len(geo_calls)}",807        )808    queried_hosts = set()809    for tc in geo_calls:810        try:811            args = json.loads(tc["function"]["arguments"])812            host = args.get("host", "")813            if not host:814                return False, f"lookup_ip_geolocation called with empty host: {args}"815            queried_hosts.add(host)816        except json.JSONDecodeError:817            return False, "lookup_ip_geolocation arguments are not valid JSON"818    expected = {"halle-eins.de", "galerie-deux.fr", "galleria-tre.it"}819    if not expected.issubset(queried_hosts):820        return (821            False,822            f"Not all domains queried. Expected {expected}, got {queried_hosts}",823        )824    if not content:825        return False, "No final summary produced"826    return True, f"All 3 domains geolocated: {sorted(queried_hosts)}"827 828 829# ---- Test 5: EV fleet expansion — stock → security → property → video ----830# Inspired by: multi-step business analysis combining finance, cybersecurity,831#              real estate and educational content.832# Anonymized: Tesla → Voltara (VLTR), Rivian → Rivex (RVXN),833#             Trenton → Halverton834 835_EV_TOOLS = [836    {837        "type": "function",838        "function": {839            "name": "get_stock_quote",840            "description": "Retrieve the latest market quote for a financial instrument by ticker symbol.",841            "parameters": {842                "type": "object",843                "properties": {844                    "symbol": {845                        "type": "string",846                        "description": "Ticker symbol (e.g. 'VLTR', 'RVXN')",847                    },848                    "interval": {849                        "type": "string",850                        "description": "Time interval: 1min, 5min, 1h, 1day, 1week",851                        "default": "1day",852                    },853                },854                "required": ["symbol"],855            },856        },857    },858    {859        "type": "function",860        "function": {861            "name": "get_security_advisories",862            "description": (863                "Fetch current cybersecurity advisories from the national security agency, "864                "covering known vulnerabilities and exploits for industrial and consumer systems."865            ),866            "parameters": {867                "type": "object",868                "properties": {869                    "keyword": {870                        "type": "string",871                        "description": "Filter advisories by keyword or product name",872                    },873                    "limit": {874                        "type": "integer",875                        "description": "Maximum number of advisories to return",876                        "default": 5,877                    },878                },879                "required": [],880            },881        },882    },883    {884        "type": "function",885        "function": {886            "name": "search_commercial_properties",887            "description": "Search for commercial properties (offices, garages, warehouses) available for rent or sale in a given city.",888            "parameters": {889                "type": "object",890                "properties": {891                    "city": {"type": "string", "description": "City name to search in"},892                    "property_type": {893                        "type": "string",894                        "description": "Type of property: office, garage, warehouse, premises",895                    },896                    "operation": {897                        "type": "string",898                        "enum": ["rent", "sale"],899                        "default": "rent",900                    },901                    "max_price": {902                        "type": "integer",903                        "description": "Maximum monthly rent or sale price",904                    },905                },906                "required": ["city", "property_type"],907            },908        },909    },910    {911        "type": "function",912        "function": {913            "name": "get_video_recommendations",914            "description": "Fetch a list of recommended videos related to a given topic or reference video.",915            "parameters": {916                "type": "object",917                "properties": {918                    "topic": {919                        "type": "string",920                        "description": "Topic or keyword to search for related videos",921                    },922                },923                "required": ["topic"],924            },925        },926    },927]928 929_STOCK_RESULT_VLTR = {930    "symbol": "VLTR",931    "company": "Voltara Inc.",932    "price": 218.45,933    "change_pct": "+2.3%",934    "market_cap": "694B",935    "currency": "USD",936}937_STOCK_RESULT_RVXN = {938    "symbol": "RVXN",939    "company": "Rivex Motors",940    "price": 12.80,941    "change_pct": "-1.1%",942    "market_cap": "11B",943    "currency": "USD",944}945_ADVISORIES_RESULT = {946    "count": 2,947    "advisories": [948        {949            "id": "ICSA-24-102-01",950            "title": "Voltara In-Vehicle Infotainment System Authentication Bypass",951            "severity": "Medium",952            "summary": "Improper authentication in the OTA update module may allow an adjacent attacker to install unsigned firmware.",953            "published": "2024-04-11",954        },955        {956            "id": "ICSA-24-085-03",957            "title": "Voltara Charging Management API Input Validation Flaw",958            "severity": "Low",959            "summary": "Insufficient input validation in the charging session API could expose internal error messages.",960            "published": "2024-03-26",961        },962    ],963}964_PROPERTIES_RESULT = {965    "city": "Halverton",966    "listings": [967        {968            "id": "HV-0041",969            "type": "garage",970            "area_sqm": 420,971            "monthly_rent": 2800,972            "ev_power_outlets": 12,973            "address": "14 Ironworks Lane, Halverton",974        },975        {976            "id": "HV-0089",977            "type": "warehouse",978            "area_sqm": 900,979            "monthly_rent": 4200,980            "ev_power_outlets": 30,981            "address": "7 Depot Road, Halverton",982        },983    ],984}985_VIDEOS_RESULT = {986    "topic": "fleet electrification",987    "recommendations": [988        {989            "title": "How to Build an EV Fleet from Scratch",990            "channel": "Fleet Future",991            "views": "182K",992        },993        {994            "title": "EV Charging Infrastructure for Commercial Fleets",995            "channel": "GreenDrive Pro",996            "views": "94K",997        },998        {999            "title": "Total Cost of Ownership: Electric vs Diesel Vans",1000            "channel": "LogisticsTech",1001            "views": "61K",1002        },1003    ],1004}1005 1006 1007def _ev_stock_mock(args):1008    symbol = args.get("symbol", "").upper()1009    if symbol == "VLTR":1010        return json.dumps(_STOCK_RESULT_VLTR)1011    if symbol == "RVXN":1012        return json.dumps(_STOCK_RESULT_RVXN)1013    return json.dumps({"error": f"Unknown symbol: {symbol}"})1014 1015 1016EV_FLEET_TEST_CASE = {1017    "name": "EV fleet expansion: stock → cybersecurity → property → videos",1018    "messages": [1019        {1020            "role": "user",1021            "content": (1022                "I'm expanding my courier business into electric vehicles and need a multi-step analysis:\n"1023                "1. Get the latest stock quote for Voltara (VLTR) and Rivex (RVXN). "1024                "If either is above $50, continue with that company.\n"1025                "2. Search for cybersecurity advisories related to that company's vehicle models "1026                "to understand any tech risks.\n"1027                "3. Find commercial garage or warehouse properties in Halverton suitable for "1028                "EV charging infrastructure.\n"1029                "4. Recommend videos on fleet electrification strategies.\n"1030                "Please work through all four steps and give me a concise summary."1031            ),1032        }1033    ],1034    "tools": _EV_TOOLS,1035    "mock_tool_responses": {1036        "get_stock_quote": _ev_stock_mock,1037        "get_security_advisories": lambda _: json.dumps(_ADVISORIES_RESULT),1038        "search_commercial_properties": lambda _: json.dumps(_PROPERTIES_RESULT),1039        "get_video_recommendations": lambda _: json.dumps(_VIDEOS_RESULT),1040    },1041    "validate": lambda tcs, content: _validate_ev(tcs, content),1042}1043 1044 1045def _validate_ev(tcs, content):1046    names = [tc["function"]["name"] for tc in tcs]1047    if not names:1048        return False, "No tool calls made"1049    # Stock quote must come first1050    if names[0] != "get_stock_quote":1051        return False, f"Expected get_stock_quote to be called first, got: {names[0]}"1052    stock_calls = [tc for tc in tcs if tc["function"]["name"] == "get_stock_quote"]1053    for tc in stock_calls:1054        try:1055            args = json.loads(tc["function"]["arguments"])1056            sym = args.get("symbol", "")1057            if not sym:1058                return False, f"get_stock_quote called with empty symbol: {args}"1059        except json.JSONDecodeError:1060            return False, "get_stock_quote arguments are not valid JSON"1061    # All four pipeline tools expected1062    required = [1063        "get_stock_quote",1064        "get_security_advisories",1065        "search_commercial_properties",1066        "get_video_recommendations",1067    ]1068    missing = [t for t in required if t not in names]1069    if missing:1070        return False, f"Missing pipeline steps: {missing}"1071    if not content:1072        return False, "No final summary produced"1073    return True, f"Full 4-step pipeline executed: {names}"1074 1075 1076# ---------------------------------------------------------------------------1077# All test cases1078# ---------------------------------------------------------------------------1079 1080ALL_TEST_CASES = [1081    AZZOO_TEST_CASE,1082    FITNESS_TEST_CASE,1083    COMMUNITY_CLASS_TEST_CASE,1084    GEO_TEST_CASE,1085    EV_FLEET_TEST_CASE,1086]1087 1088 1089# ---------------------------------------------------------------------------1090# Entry point1091# ---------------------------------------------------------------------------1092 1093 1094def main():1095    parser = argparse.ArgumentParser(1096        description="Test llama-server tool-calling capability."1097    )1098    parser.add_argument("--host", default="localhost")1099    parser.add_argument("--port", default=8080, type=int)1100    parser.add_argument(1101        "--no-stream", action="store_true", help="Disable streaming mode tests"1102    )1103    parser.add_argument(1104        "--stream-only", action="store_true", help="Only run streaming mode tests"1105    )1106    parser.add_argument(1107        "--force-tools", action="store_true", help="Change tool mode to forced instead of auto"1108    )1109    parser.add_argument(1110        "--test",1111        help="Run only the test whose name contains this substring (case-insensitive)",1112    )1113    args = parser.parse_args()1114 1115    url = f"http://{args.host}:{args.port}/v1/chat/completions"1116    print_info(f"Testing server at {url}")1117 1118    modes = []1119    force_tools = False1120    if not args.stream_only:1121        modes.append(False)1122    if not args.no_stream:1123        modes.append(True)1124    if args.force_tools:1125        force_tools = True1126 1127    cases: list[dict] = ALL_TEST_CASES1128    if args.test:1129        name_filter = args.test.lower()1130        cases = [c for c in cases if name_filter in str(c["name"]).lower()]1131        if not cases:1132            print_fail(f"No test cases matched '{args.test}'")1133            sys.exit(1)1134 1135    total = 01136    passed = 01137    for stream in modes:1138        for case in cases:1139            total += 11140            if run_test(url, case, stream=stream, force_tools=force_tools):1141                passed += 11142 1143    color = GREEN if passed == total else RED1144    _print(f"\n{BOLD}{color}{'─'*60}{RESET}")1145    _print(f"{BOLD}{color}  Results: {passed}/{total} passed{RESET}")1146    _print(f"{BOLD}{color}{'─'*60}{RESET}\n")1147    sys.exit(0 if passed == total else 1)1148 1149 1150if __name__ == "__main__":1151    main()1152 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai