KBaba7/llama.cpp
0
1import pytest2from utils import *3 4server: ServerProcess5 6TIMEOUT_SERVER_START = 15*607TIMEOUT_HTTP_REQUEST = 608 9@pytest.fixture(autouse=True)10def create_server():11 global server12 server = ServerPreset.tinyllama2()13 server.model_alias = "tinyllama-2-tool-call"14 server.server_port = 808115 16 17TEST_TOOL = {18 "type":"function",19 "function": {20 "name": "test",21 "description": "",22 "parameters": {23 "type": "object",24 "properties": {25 "success": {"type": "boolean", "const": True},26 },27 "required": ["success"]28 }29 }30}31 32PYTHON_TOOL = {33 "type": "function",34 "function": {35 "name": "python",36 "description": "Runs code in an ipython interpreter and returns the result of the execution after 60 seconds.",37 "parameters": {38 "type": "object",39 "properties": {40 "code": {41 "type": "string",42 "description": "The code to run in the ipython interpreter."43 }44 },45 "required": ["code"]46 }47 }48}49 50WEATHER_TOOL = {51 "type":"function",52 "function":{53 "name":"get_current_weather",54 "description":"Get the current weather in a given location",55 "parameters":{56 "type":"object",57 "properties":{58 "location":{59 "type":"string",60 "description":"The city and country/state, e.g. 'San Francisco, CA', or 'Paris, France'"61 }62 },63 "required":["location"]64 }65 }66}67 68 69def do_test_completion_with_required_tool_tiny(template_name: str, tool: dict, argument_key: str | None):70 global server71 n_predict = 51272 # server = ServerPreset.stories15m_moe()73 server.jinja = True74 server.n_predict = n_predict75 server.chat_template_file = f'../../../models/templates/{template_name}.jinja'76 server.start(timeout_seconds=TIMEOUT_SERVER_START)77 res = server.make_request("POST", "/chat/completions", data={78 "max_tokens": n_predict,79 "messages": [80 {"role": "system", "content": "You are a coding assistant."},81 {"role": "user", "content": "Write an example"},82 ],83 "tool_choice": "required",84 "tools": [tool],85 "parallel_tool_calls": False,86 "temperature": 0.0,87 "top_k": 1,88 "top_p": 1.0,89 })90 assert res.status_code == 200, f"Expected status code 200, got {res.status_code}"91 choice = res.body["choices"][0]92 tool_calls = choice["message"].get("tool_calls")93 assert tool_calls and len(tool_calls) == 1, f'Expected 1 tool call in {choice["message"]}'94 tool_call = tool_calls[0]95 expected_function_name = "python" if tool["type"] == "code_interpreter" else tool["function"]["name"]96 assert expected_function_name == tool_call["function"]["name"]97 actual_arguments = tool_call["function"]["arguments"]98 assert isinstance(actual_arguments, str)99 if argument_key is not None:100 actual_arguments = json.loads(actual_arguments)101 assert argument_key in actual_arguments, f"tool arguments: {json.dumps(actual_arguments)}, expected: {argument_key}"102 103 104@pytest.mark.parametrize("template_name,tool,argument_key", [105 ("google-gemma-2-2b-it", TEST_TOOL, "success"),106 ("meta-llama-Llama-3.3-70B-Instruct", TEST_TOOL, "success"),107 ("meta-llama-Llama-3.3-70B-Instruct", PYTHON_TOOL, "code"),108])109def test_completion_with_required_tool_tiny_fast(template_name: str, tool: dict, argument_key: str | None):110 do_test_completion_with_required_tool_tiny(template_name, tool, argument_key)111 112 113@pytest.mark.slow114@pytest.mark.parametrize("template_name,tool,argument_key", [115 ("meta-llama-Llama-3.1-8B-Instruct", TEST_TOOL, "success"),116 ("meta-llama-Llama-3.1-8B-Instruct", PYTHON_TOOL, "code"),117 ("meetkai-functionary-medium-v3.1", TEST_TOOL, "success"),118 ("meetkai-functionary-medium-v3.1", PYTHON_TOOL, "code"),119 ("meetkai-functionary-medium-v3.2", TEST_TOOL, "success"),120 ("meetkai-functionary-medium-v3.2", PYTHON_TOOL, "code"),121 ("NousResearch-Hermes-2-Pro-Llama-3-8B-tool_use", TEST_TOOL, "success"),122 ("NousResearch-Hermes-2-Pro-Llama-3-8B-tool_use", PYTHON_TOOL, "code"),123 ("meta-llama-Llama-3.2-3B-Instruct", TEST_TOOL, "success"),124 ("meta-llama-Llama-3.2-3B-Instruct", PYTHON_TOOL, "code"),125 ("mistralai-Mistral-Nemo-Instruct-2407", TEST_TOOL, "success"),126 ("mistralai-Mistral-Nemo-Instruct-2407", PYTHON_TOOL, "code"),127 ("NousResearch-Hermes-3-Llama-3.1-8B-tool_use", TEST_TOOL, "success"),128 ("NousResearch-Hermes-3-Llama-3.1-8B-tool_use", PYTHON_TOOL, "code"),129 ("deepseek-ai-DeepSeek-R1-Distill-Llama-8B", TEST_TOOL, "success"),130 ("deepseek-ai-DeepSeek-R1-Distill-Llama-8B", PYTHON_TOOL, "code"),131 ("fireworks-ai-llama-3-firefunction-v2", TEST_TOOL, "success"),132 ("fireworks-ai-llama-3-firefunction-v2", PYTHON_TOOL, "code"),133])134def test_completion_with_required_tool_tiny_slow(template_name: str, tool: dict, argument_key: str | None):135 do_test_completion_with_required_tool_tiny(template_name, tool, argument_key)136 137 138@pytest.mark.slow139@pytest.mark.parametrize("tool,argument_key,hf_repo,template_override", [140 (TEST_TOOL, "success", "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF:Q4_K_M", None),141 (PYTHON_TOOL, "code", "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF:Q4_K_M", None),142 (PYTHON_TOOL, "code", "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF:Q4_K_M", "chatml"),143 144 # Note: gemma-2-2b-it knows itself as "model", not "assistant", so we don't test the ill-suited chatml on it.145 (TEST_TOOL, "success", "bartowski/gemma-2-2b-it-GGUF:Q4_K_M", None),146 (PYTHON_TOOL, "code", "bartowski/gemma-2-2b-it-GGUF:Q4_K_M", None),147 148 (TEST_TOOL, "success", "bartowski/Phi-3.5-mini-instruct-GGUF:Q4_K_M", None),149 (PYTHON_TOOL, "code", "bartowski/Phi-3.5-mini-instruct-GGUF:Q4_K_M", None),150 (PYTHON_TOOL, "code", "bartowski/Phi-3.5-mini-instruct-GGUF:Q4_K_M", "chatml"),151 152 (TEST_TOOL, "success", "bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M", None),153 (PYTHON_TOOL, "code", "bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M", None),154 (PYTHON_TOOL, "code", "bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M", "chatml"),155 156 (TEST_TOOL, "success", "bartowski/Hermes-2-Pro-Llama-3-8B-GGUF:Q4_K_M", ("NousResearch/Hermes-2-Pro-Llama-3-8B", "tool_use")),157 (PYTHON_TOOL, "code", "bartowski/Hermes-2-Pro-Llama-3-8B-GGUF:Q4_K_M", ("NousResearch/Hermes-2-Pro-Llama-3-8B", "tool_use")),158 (PYTHON_TOOL, "code", "bartowski/Hermes-2-Pro-Llama-3-8B-GGUF:Q4_K_M", "chatml"),159 160 (TEST_TOOL, "success", "bartowski/Hermes-3-Llama-3.1-8B-GGUF:Q4_K_M", ("NousResearch/Hermes-3-Llama-3.1-8B", "tool_use")),161 (PYTHON_TOOL, "code", "bartowski/Hermes-3-Llama-3.1-8B-GGUF:Q4_K_M", ("NousResearch/Hermes-3-Llama-3.1-8B", "tool_use")),162 (PYTHON_TOOL, "code", "bartowski/Hermes-3-Llama-3.1-8B-GGUF:Q4_K_M", "chatml"),163 164 (TEST_TOOL, "success", "bartowski/Mistral-Nemo-Instruct-2407-GGUF:Q4_K_M", None),165 (PYTHON_TOOL, "code", "bartowski/Mistral-Nemo-Instruct-2407-GGUF:Q4_K_M", None),166 (PYTHON_TOOL, "code", "bartowski/Mistral-Nemo-Instruct-2407-GGUF:Q4_K_M", "chatml"),167 168 (TEST_TOOL, "success", "bartowski/functionary-small-v3.2-GGUF:Q4_K_M", ("meetkai/functionary-medium-v3.2", None)),169 (PYTHON_TOOL, "code", "bartowski/functionary-small-v3.2-GGUF:Q4_K_M", ("meetkai/functionary-medium-v3.2", None)),170 (PYTHON_TOOL, "code", "bartowski/functionary-small-v3.2-GGUF:Q4_K_M", "chatml"),171 172 (TEST_TOOL, "success", "bartowski/Llama-3.2-3B-Instruct-GGUF:Q4_K_M", ("meta-llama/Llama-3.2-3B-Instruct", None)),173 (PYTHON_TOOL, "code", "bartowski/Llama-3.2-3B-Instruct-GGUF:Q4_K_M", ("meta-llama/Llama-3.2-3B-Instruct", None)),174 (PYTHON_TOOL, "code", "bartowski/Llama-3.2-3B-Instruct-GGUF:Q4_K_M", "chatml"),175 176 (TEST_TOOL, "success", "bartowski/Llama-3.2-1B-Instruct-GGUF:Q4_K_M", ("meta-llama/Llama-3.2-3B-Instruct", None)),177 (PYTHON_TOOL, "code", "bartowski/Llama-3.2-1B-Instruct-GGUF:Q4_K_M", ("meta-llama/Llama-3.2-3B-Instruct", None)),178 (PYTHON_TOOL, "code", "bartowski/Llama-3.2-1B-Instruct-GGUF:Q4_K_M", "chatml"),179 # TODO: fix these180 # (TEST_TOOL, "success", "bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF:Q4_K_M", None),181 # (PYTHON_TOOL, "code", "bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF:Q4_K_M", None),182])183def test_completion_with_required_tool_real_model(tool: dict, argument_key: str | None, hf_repo: str, template_override: str | Tuple[str, str | None] | None):184 global server185 n_predict = 512186 server.n_slots = 1187 server.jinja = True188 server.n_ctx = 8192189 server.n_predict = n_predict190 server.model_hf_repo = hf_repo191 server.model_hf_file = None192 if isinstance(template_override, tuple):193 (template_hf_repo, template_variant) = template_override194 server.chat_template_file = f"../../../models/templates/{template_hf_repo.replace('/', '-') + ('-' + template_variant if template_variant else '')}.jinja"195 assert os.path.exists(server.chat_template_file), f"Template file {server.chat_template_file} does not exist. Run `python scripts/get_chat_template.py {template_hf_repo} {template_variant} > {server.chat_template_file}` to download the template."196 elif isinstance(template_override, str):197 server.chat_template = template_override198 server.start(timeout_seconds=TIMEOUT_SERVER_START)199 res = server.make_request("POST", "/chat/completions", data={200 "max_tokens": n_predict,201 "messages": [202 {"role": "system", "content": "You are a coding assistant."},203 {"role": "user", "content": "Write an example"},204 ],205 "tool_choice": "required",206 "tools": [tool],207 "parallel_tool_calls": False,208 "temperature": 0.0,209 "top_k": 1,210 "top_p": 1.0,211 }, timeout=TIMEOUT_HTTP_REQUEST)212 assert res.status_code == 200, f"Expected status code 200, got {res.status_code}"213 choice = res.body["choices"][0]214 tool_calls = choice["message"].get("tool_calls")215 assert tool_calls and len(tool_calls) == 1, f'Expected 1 tool call in {choice["message"]}'216 tool_call = tool_calls[0]217 expected_function_name = "python" if tool["type"] == "code_interpreter" else tool["function"]["name"]218 assert expected_function_name == tool_call["function"]["name"]219 actual_arguments = tool_call["function"]["arguments"]220 assert isinstance(actual_arguments, str)221 if argument_key is not None:222 actual_arguments = json.loads(actual_arguments)223 assert argument_key in actual_arguments, f"tool arguments: {json.dumps(actual_arguments)}, expected: {argument_key}"224 225 226def do_test_completion_without_tool_call(template_name: str, n_predict: int, tools: list[dict], tool_choice: str | None):227 global server228 server.jinja = True229 server.n_predict = n_predict230 server.chat_template_file = f'../../../models/templates/{template_name}.jinja'231 server.start(timeout_seconds=TIMEOUT_SERVER_START)232 res = server.make_request("POST", "/chat/completions", data={233 "max_tokens": n_predict,234 "messages": [235 {"role": "system", "content": "You are a coding assistant."},236 {"role": "user", "content": "say hello world with python"},237 ],238 "tools": tools if tools else None,239 "tool_choice": tool_choice,240 "temperature": 0.0,241 "top_k": 1,242 "top_p": 1.0,243 }, timeout=TIMEOUT_HTTP_REQUEST)244 assert res.status_code == 200, f"Expected status code 200, got {res.status_code}"245 choice = res.body["choices"][0]246 assert choice["message"].get("tool_calls") is None, f'Expected no tool call in {choice["message"]}'247 248 249@pytest.mark.parametrize("template_name,n_predict,tools,tool_choice", [250 ("meta-llama-Llama-3.3-70B-Instruct", 128, [], None),251 ("meta-llama-Llama-3.3-70B-Instruct", 128, [TEST_TOOL], None),252 ("meta-llama-Llama-3.3-70B-Instruct", 128, [PYTHON_TOOL], 'none'),253])254def test_completion_without_tool_call_fast(template_name: str, n_predict: int, tools: list[dict], tool_choice: str | None):255 do_test_completion_without_tool_call(template_name, n_predict, tools, tool_choice)256 257 258@pytest.mark.slow259@pytest.mark.parametrize("template_name,n_predict,tools,tool_choice", [260 ("meetkai-functionary-medium-v3.2", 256, [], None),261 ("meetkai-functionary-medium-v3.2", 256, [TEST_TOOL], None),262 ("meetkai-functionary-medium-v3.2", 256, [PYTHON_TOOL], 'none'),263 ("meetkai-functionary-medium-v3.1", 256, [], None),264 ("meetkai-functionary-medium-v3.1", 256, [TEST_TOOL], None),265 ("meetkai-functionary-medium-v3.1", 256, [PYTHON_TOOL], 'none'),266 ("meta-llama-Llama-3.2-3B-Instruct", 256, [], None),267 ("meta-llama-Llama-3.2-3B-Instruct", 256, [TEST_TOOL], None),268 ("meta-llama-Llama-3.2-3B-Instruct", 256, [PYTHON_TOOL], 'none'),269])270def test_completion_without_tool_call_slow(template_name: str, n_predict: int, tools: list[dict], tool_choice: str | None):271 do_test_completion_without_tool_call(template_name, n_predict, tools, tool_choice)272 273 274@pytest.mark.slow275@pytest.mark.parametrize("hf_repo,template_override", [276 ("bartowski/c4ai-command-r7b-12-2024-GGUF:Q4_K_M", ("CohereForAI/c4ai-command-r7b-12-2024", "tool_use")),277 ("bartowski/Meta-Llama-3.1-8B-Instruct-GGUF:Q4_K_M", None),278 ("bartowski/Meta-Llama-3.1-8B-Instruct-GGUF:Q4_K_M", "chatml"),279 280 ("bartowski/Phi-3.5-mini-instruct-GGUF:Q4_K_M", None),281 ("bartowski/Phi-3.5-mini-instruct-GGUF:Q4_K_M", "chatml"),282 283 ("bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M", None),284 ("bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M", "chatml"),285 286 ("bartowski/Hermes-2-Pro-Llama-3-8B-GGUF:Q4_K_M", ("NousResearch/Hermes-2-Pro-Llama-3-8B", "tool_use")),287 ("bartowski/Hermes-2-Pro-Llama-3-8B-GGUF:Q4_K_M", "chatml"),288 289 ("bartowski/Hermes-3-Llama-3.1-8B-GGUF:Q4_K_M", ("NousResearch/Hermes-3-Llama-3.1-8B", "tool_use")),290 ("bartowski/Hermes-3-Llama-3.1-8B-GGUF:Q4_K_M", "chatml"),291 292 ("bartowski/Mistral-Nemo-Instruct-2407-GGUF:Q4_K_M", None),293 ("bartowski/Mistral-Nemo-Instruct-2407-GGUF:Q4_K_M", "chatml"),294 295 ("bartowski/functionary-small-v3.2-GGUF:Q8_0", ("meetkai/functionary-medium-v3.2", None)),296 ("bartowski/functionary-small-v3.2-GGUF:Q8_0", "chatml"),297 298 ("bartowski/Llama-3.2-3B-Instruct-GGUF:Q4_K_M", ("meta-llama/Llama-3.2-3B-Instruct", None)),299 ("bartowski/Llama-3.2-3B-Instruct-GGUF:Q4_K_M", "chatml"),300 301 # Note: gemma-2-2b-it knows itself as "model", not "assistant", so we don't test the ill-suited chatml on it.302 ("bartowski/gemma-2-2b-it-GGUF:Q4_K_M", None),303 304 # ("bartowski/Llama-3.2-1B-Instruct-GGUF:Q4_K_M", ("meta-llama/Llama-3.2-3B-Instruct", None)),305 # ("bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF:Q4_K_M", None),306])307def test_weather(hf_repo: str, template_override: Tuple[str, str | None] | None):308 global server309 n_predict = 512310 server.n_slots = 1311 server.jinja = True312 server.n_ctx = 8192313 server.n_predict = n_predict314 server.model_hf_repo = hf_repo315 server.model_hf_file = None316 if isinstance(template_override, tuple):317 (template_hf_repo, template_variant) = template_override318 server.chat_template_file = f"../../../models/templates/{template_hf_repo.replace('/', '-') + ('-' + template_variant if template_variant else '')}.jinja"319 assert os.path.exists(server.chat_template_file), f"Template file {server.chat_template_file} does not exist. Run `python scripts/get_chat_template.py {template_hf_repo} {template_variant} > {server.chat_template_file}` to download the template."320 elif isinstance(template_override, str):321 server.chat_template = template_override322 server.start(timeout_seconds=TIMEOUT_SERVER_START)323 res = server.make_request("POST", "/chat/completions", data={324 "max_tokens": n_predict,325 "messages": [326 {"role": "user", "content": "What is the weather in Istanbul?"},327 ],328 "tools": [WEATHER_TOOL],329 }, timeout=TIMEOUT_HTTP_REQUEST)330 assert res.status_code == 200, f"Expected status code 200, got {res.status_code}"331 choice = res.body["choices"][0]332 tool_calls = choice["message"].get("tool_calls")333 assert tool_calls and len(tool_calls) == 1, f'Expected 1 tool call in {choice["message"]}'334 tool_call = tool_calls[0]335 assert tool_call["function"]["name"] == WEATHER_TOOL["function"]["name"]336 actual_arguments = json.loads(tool_call["function"]["arguments"])337 assert 'location' in actual_arguments, f"location not found in {json.dumps(actual_arguments)}"338 location = actual_arguments["location"]339 assert isinstance(location, str), f"Expected location to be a string, got {type(location)}: {json.dumps(location)}"340 assert re.match('^Istanbul(, (TR|Turkey|Türkiye))?$', location), f'Expected Istanbul for location, got {location}'341 342 343@pytest.mark.slow344@pytest.mark.parametrize("expected_arguments_override,hf_repo,template_override", [345 (None, "bartowski/Phi-3.5-mini-instruct-GGUF:Q4_K_M", None),346 (None, "bartowski/Phi-3.5-mini-instruct-GGUF:Q4_K_M", "chatml"),347 348 (None, "bartowski/functionary-small-v3.2-GGUF:Q8_0", ("meetkai-functionary-medium-v3.2", None)),349 (None, "bartowski/functionary-small-v3.2-GGUF:Q8_0", "chatml"),350 351 (None, "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF:Q4_K_M", None),352 ('{"code":"print("}', "bartowski/Meta-Llama-3.1-8B-Instruct-GGUF:Q4_K_M", "chatml"),353 354 ('{"code":"print("}', "bartowski/Llama-3.2-1B-Instruct-GGUF:Q4_K_M", ("meta-llama-Llama-3.2-3B-Instruct", None)),355 (None, "bartowski/Llama-3.2-1B-Instruct-GGUF:Q4_K_M", "chatml"),356 357 ('{"code":"print("}', "bartowski/Llama-3.2-3B-Instruct-GGUF:Q4_K_M", ("meta-llama-Llama-3.2-3B-Instruct", None)),358 ('{"code":"print("}', "bartowski/Llama-3.2-3B-Instruct-GGUF:Q4_K_M", "chatml"),359 360 (None, "bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M", None),361 (None, "bartowski/Qwen2.5-7B-Instruct-GGUF:Q4_K_M", "chatml"),362 363 (None, "bartowski/Hermes-2-Pro-Llama-3-8B-GGUF:Q4_K_M", ("NousResearch/Hermes-2-Pro-Llama-3-8B", "tool_use")),364 (None, "bartowski/Hermes-2-Pro-Llama-3-8B-GGUF:Q4_K_M", "chatml"),365 366 (None, "bartowski/Hermes-3-Llama-3.1-8B-GGUF:Q4_K_M", ("NousResearch-Hermes-3-Llama-3.1-8B", "tool_use")),367 (None, "bartowski/Hermes-3-Llama-3.1-8B-GGUF:Q4_K_M", "chatml"),368 369 (None, "bartowski/Mistral-Nemo-Instruct-2407-GGUF:Q4_K_M", None),370 (None, "bartowski/Mistral-Nemo-Instruct-2407-GGUF:Q4_K_M", "chatml"),371 372 # Note: gemma-2-2b-it knows itself as "model", not "assistant", so we don't test the ill-suited chatml on it.373 (None, "bartowski/gemma-2-2b-it-GGUF:Q4_K_M", None),374 375 # (None, "bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF:Q4_K_M", None),376])377def test_hello_world_tool_call(expected_arguments_override: str | None, hf_repo: str, template_override: str | Tuple[str, str | None] | None):378 global server379 server.n_slots = 1380 server.jinja = True381 server.n_ctx = 8192382 server.n_predict = 128383 server.model_hf_repo = hf_repo384 server.model_hf_file = None385 if isinstance(template_override, tuple):386 (template_hf_repo, template_variant) = template_override387 server.chat_template_file = f"../../../models/templates/{template_hf_repo.replace('/', '-') + ('-' + template_variant if template_variant else '')}.jinja"388 assert os.path.exists(server.chat_template_file), f"Template file {server.chat_template_file} does not exist. Run `python scripts/get_chat_template.py {template_hf_repo} {template_variant} > {server.chat_template_file}` to download the template."389 elif isinstance(template_override, str):390 server.chat_template = template_override391 server.start(timeout_seconds=TIMEOUT_SERVER_START)392 res = server.make_request("POST", "/chat/completions", data={393 "max_tokens": 256,394 "messages": [395 {"role": "system", "content": "You are a coding assistant."},396 {"role": "user", "content": "say hello world with python"},397 ],398 "tools": [PYTHON_TOOL],399 # Note: without these greedy params, Functionary v3.2 writes `def hello_world():\n print("Hello, World!")\nhello_world()` which is correct but a pain to test.400 "temperature": 0.0,401 "top_k": 1,402 "top_p": 1.0,403 }, timeout=TIMEOUT_HTTP_REQUEST)404 assert res.status_code == 200, f"Expected status code 200, got {res.status_code}"405 choice = res.body["choices"][0]406 tool_calls = choice["message"].get("tool_calls")407 assert tool_calls and len(tool_calls) == 1, f'Expected 1 tool call in {choice["message"]}'408 tool_call = tool_calls[0]409 assert tool_call["function"]["name"] == PYTHON_TOOL["function"]["name"]410 actual_arguments = tool_call["function"]["arguments"]411 if expected_arguments_override is not None:412 assert actual_arguments == expected_arguments_override413 else:414 actual_arguments = json.loads(actual_arguments)415 assert 'code' in actual_arguments, f"code not found in {json.dumps(actual_arguments)}"416 code = actual_arguments["code"]417 assert isinstance(code, str), f"Expected code to be a string, got {type(code)}: {json.dumps(code)}"418 assert re.match(r'''print\(("[Hh]ello,? [Ww]orld!?"|'[Hh]ello,? [Ww]orld!?')\)''', code), f'Expected hello world, got {code}'419 