Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#!/usr/bin/env python32import pytest3import base644import requests5 6from utils import *7 8server: ServerProcess9 10 11def get_test_image_base64() -> str:12 """Get a test image in base64 format"""13 # Use the same test image as test_vision_api.py14 IMG_URL = "https://huggingface.co/ggml-org/tinygemma3-GGUF/resolve/main/test/11_truck.png"15 response = requests.get(IMG_URL)16 response.raise_for_status()17 return base64.b64encode(response.content).decode("utf-8")18 19@pytest.fixture(autouse=True)20def create_server():21 global server22 server = ServerPreset.tinyllama2()23 server.model_alias = "tinyllama-2-anthropic"24 server.server_port = 808225 server.n_slots = 126 server.n_ctx = 819227 server.n_batch = 204828 29 30@pytest.fixture31def vision_server():32 """Separate fixture for vision tests that require multimodal support"""33 global server34 server = ServerPreset.tinygemma3()35 server.offline = False # Allow downloading the model36 server.model_alias = "tinygemma3-anthropic"37 server.server_port = 8083 # Different port to avoid conflicts38 server.n_slots = 139 return server40 41 42# Basic message tests43 44def test_anthropic_messages_basic():45 """Test basic Anthropic messages endpoint"""46 server.start()47 48 res = server.make_request("POST", "/v1/messages", data={49 "model": "test",50 "max_tokens": 50,51 "messages": [52 {"role": "user", "content": "Say hello"}53 ]54 })55 56 assert res.status_code == 200, f"Expected 200, got {res.status_code}"57 assert res.body["type"] == "message", f"Expected type 'message', got {res.body.get('type')}"58 assert res.body["role"] == "assistant", f"Expected role 'assistant', got {res.body.get('role')}"59 assert "content" in res.body, "Missing 'content' field"60 assert isinstance(res.body["content"], list), "Content should be an array"61 assert len(res.body["content"]) > 0, "Content array should not be empty"62 assert res.body["content"][0]["type"] == "text", "First content block should be text"63 assert "text" in res.body["content"][0], "Text content block missing 'text' field"64 assert res.body["stop_reason"] in ["end_turn", "max_tokens"], f"Invalid stop_reason: {res.body.get('stop_reason')}"65 assert "usage" in res.body, "Missing 'usage' field"66 assert "cache_read_input_tokens" in res.body["usage"], "Missing usage.cache_read_input_tokens"67 assert "input_tokens" in res.body["usage"], "Missing usage.input_tokens"68 assert "output_tokens" in res.body["usage"], "Missing usage.output_tokens"69 assert isinstance(res.body["usage"]["cache_read_input_tokens"], int), "cache_read_input_tokens should be integer"70 assert isinstance(res.body["usage"]["input_tokens"], int), "input_tokens should be integer"71 assert isinstance(res.body["usage"]["output_tokens"], int), "output_tokens should be integer"72 assert res.body["usage"]["output_tokens"] > 0, "Should have generated some tokens"73 # Anthropic API should NOT include timings74 assert "timings" not in res.body, "Anthropic API should not include timings field"75 76 77def test_anthropic_messages_with_system():78 """Test messages with system prompt"""79 server.start()80 81 res = server.make_request("POST", "/v1/messages", data={82 "model": "test",83 "max_tokens": 50,84 "system": "You are a helpful assistant.",85 "messages": [86 {"role": "user", "content": "Hello"}87 ]88 })89 90 assert res.status_code == 20091 assert res.body["type"] == "message"92 assert len(res.body["content"]) > 093 94 95def test_anthropic_messages_multipart_content():96 """Test messages with multipart content blocks"""97 server.start()98 99 res = server.make_request("POST", "/v1/messages", data={100 "model": "test",101 "max_tokens": 50,102 "messages": [103 {104 "role": "user",105 "content": [106 {"type": "text", "text": "What is"},107 {"type": "text", "text": " the answer?"}108 ]109 }110 ]111 })112 113 assert res.status_code == 200114 assert res.body["type"] == "message"115 116 117def test_anthropic_messages_conversation():118 """Test multi-turn conversation"""119 server.start()120 121 res = server.make_request("POST", "/v1/messages", data={122 "model": "test",123 "max_tokens": 50,124 "messages": [125 {"role": "user", "content": "Hello"},126 {"role": "assistant", "content": "Hi there!"},127 {"role": "user", "content": "How are you?"}128 ]129 })130 131 assert res.status_code == 200132 assert res.body["type"] == "message"133 134 135# Streaming tests136 137def test_anthropic_messages_streaming():138 """Test streaming messages"""139 server.start()140 141 res = server.make_stream_request("POST", "/v1/messages", data={142 "model": "test",143 "max_tokens": 30,144 "messages": [145 {"role": "user", "content": "Say hello"}146 ],147 "stream": True148 })149 150 events = []151 for data in res:152 # Each event should have type and other fields153 assert "type" in data, f"Missing 'type' in event: {data}"154 events.append(data)155 156 # Verify event sequence157 event_types = [e["type"] for e in events]158 assert "message_start" in event_types, "Missing message_start event"159 assert "content_block_start" in event_types, "Missing content_block_start event"160 assert "content_block_delta" in event_types, "Missing content_block_delta event"161 assert "content_block_stop" in event_types, "Missing content_block_stop event"162 assert "message_delta" in event_types, "Missing message_delta event"163 assert "message_stop" in event_types, "Missing message_stop event"164 165 # Check message_start structure166 message_start = next(e for e in events if e["type"] == "message_start")167 assert "message" in message_start, "message_start missing 'message' field"168 assert message_start["message"]["type"] == "message"169 assert message_start["message"]["role"] == "assistant"170 assert message_start["message"]["content"] == []171 assert "usage" in message_start["message"]172 assert message_start["message"]["usage"]["input_tokens"] > 0173 174 # Check content_block_start175 block_start = next(e for e in events if e["type"] == "content_block_start")176 assert "index" in block_start, "content_block_start missing 'index'"177 assert block_start["index"] == 0, "First content block should be at index 0"178 assert "content_block" in block_start179 assert block_start["content_block"]["type"] == "text"180 181 # Check content_block_delta182 deltas = [e for e in events if e["type"] == "content_block_delta"]183 assert len(deltas) > 0, "Should have at least one content_block_delta"184 for delta in deltas:185 assert "index" in delta186 assert "delta" in delta187 assert delta["delta"]["type"] == "text_delta"188 assert "text" in delta["delta"]189 190 # Check content_block_stop191 block_stop = next(e for e in events if e["type"] == "content_block_stop")192 assert "index" in block_stop193 assert block_stop["index"] == 0194 195 # Check message_delta196 message_delta = next(e for e in events if e["type"] == "message_delta")197 assert "delta" in message_delta198 assert "stop_reason" in message_delta["delta"]199 assert message_delta["delta"]["stop_reason"] in ["end_turn", "max_tokens"]200 assert "usage" in message_delta201 assert message_delta["usage"]["output_tokens"] > 0202 203 # Check message_stop204 message_stop = next(e for e in events if e["type"] == "message_stop")205 # message_stop should NOT have timings for Anthropic API206 assert "timings" not in message_stop, "Anthropic streaming should not include timings"207 208 209# Token counting tests210 211def test_anthropic_count_tokens():212 """Test token counting endpoint"""213 server.start()214 215 res = server.make_request("POST", "/v1/messages/count_tokens", data={216 "model": "test",217 "messages": [218 {"role": "user", "content": "Hello world"}219 ]220 })221 222 assert res.status_code == 200223 assert "input_tokens" in res.body224 assert isinstance(res.body["input_tokens"], int)225 assert res.body["input_tokens"] > 0226 # Should only have input_tokens, no other fields227 assert "output_tokens" not in res.body228 229 230def test_anthropic_count_tokens_with_system():231 """Test token counting with system prompt"""232 server.start()233 234 res = server.make_request("POST", "/v1/messages/count_tokens", data={235 "model": "test",236 "system": "You are a helpful assistant.",237 "messages": [238 {"role": "user", "content": "Hello"}239 ]240 })241 242 assert res.status_code == 200243 assert res.body["input_tokens"] > 0244 245 246def test_anthropic_count_tokens_no_max_tokens():247 """Test that count_tokens doesn't require max_tokens"""248 server.start()249 250 # max_tokens is NOT required for count_tokens251 res = server.make_request("POST", "/v1/messages/count_tokens", data={252 "model": "test",253 "messages": [254 {"role": "user", "content": "Hello"}255 ]256 })257 258 assert res.status_code == 200259 assert "input_tokens" in res.body260 261 262# Tool use tests263 264def test_anthropic_tool_use_basic():265 """Test basic tool use"""266 server.jinja = True267 server.start()268 269 res = server.make_request("POST", "/v1/messages", data={270 "model": "test",271 "max_tokens": 200,272 "tools": [{273 "name": "get_weather",274 "description": "Get the current weather in a location",275 "input_schema": {276 "type": "object",277 "properties": {278 "location": {279 "type": "string",280 "description": "City name"281 }282 },283 "required": ["location"]284 }285 }],286 "messages": [287 {"role": "user", "content": "What's the weather in Paris?"}288 ]289 })290 291 assert res.status_code == 200292 assert res.body["type"] == "message"293 assert len(res.body["content"]) > 0294 295 # Check if model used the tool (it might not always, depending on the model)296 content_types = [block.get("type") for block in res.body["content"]]297 298 if "tool_use" in content_types:299 # Model used the tool300 assert res.body["stop_reason"] == "tool_use"301 302 # Find the tool_use block303 tool_block = next(b for b in res.body["content"] if b.get("type") == "tool_use")304 assert "id" in tool_block305 assert "name" in tool_block306 assert tool_block["name"] == "get_weather"307 assert "input" in tool_block308 assert isinstance(tool_block["input"], dict)309 310 311def test_anthropic_tool_result():312 """Test sending tool results back313 314 This test verifies that tool_result blocks are properly converted to315 role="tool" messages internally. Without proper conversion, this would316 fail with a 500 error: "unsupported content[].type" because tool_result317 blocks would remain in the user message content array.318 """319 server.jinja = True320 server.start()321 322 res = server.make_request("POST", "/v1/messages", data={323 "model": "test",324 "max_tokens": 100,325 "messages": [326 {"role": "user", "content": "What's the weather?"},327 {328 "role": "assistant",329 "content": [330 {331 "type": "tool_use",332 "id": "test123",333 "name": "get_weather",334 "input": {"location": "Paris"}335 }336 ]337 },338 {339 "role": "user",340 "content": [341 {342 "type": "tool_result",343 "tool_use_id": "test123",344 "content": "The weather is sunny, 25°C"345 }346 ]347 }348 ]349 })350 351 # This would be 500 with the old bug where tool_result blocks weren't converted352 assert res.status_code == 200353 assert res.body["type"] == "message"354 # Model should respond to the tool result355 assert len(res.body["content"]) > 0356 assert res.body["content"][0]["type"] == "text"357 358 359def test_anthropic_tool_result_with_text():360 """Test tool result mixed with text content361 362 This tests the edge case where a user message contains both text and363 tool_result blocks. The server must properly split these into separate364 messages: a user message with text, followed by tool messages.365 Without proper handling, this would fail with 500: "unsupported content[].type"366 """367 server.jinja = True368 server.start()369 370 res = server.make_request("POST", "/v1/messages", data={371 "model": "test",372 "max_tokens": 100,373 "messages": [374 {"role": "user", "content": "What's the weather?"},375 {376 "role": "assistant",377 "content": [378 {379 "type": "tool_use",380 "id": "tool_1",381 "name": "get_weather",382 "input": {"location": "Paris"}383 }384 ]385 },386 {387 "role": "user",388 "content": [389 {"type": "text", "text": "Here are the results:"},390 {391 "type": "tool_result",392 "tool_use_id": "tool_1",393 "content": "Sunny, 25°C"394 }395 ]396 }397 ]398 })399 400 assert res.status_code == 200401 assert res.body["type"] == "message"402 assert len(res.body["content"]) > 0403 404 405def test_anthropic_tool_result_with_image():406 """Test tool result containing mixed text and image blocks407 408 Verifies that image blocks inside Anthropic tool_result content are409 properly converted to OpenAI image_url format rather than being410 silently dropped. With a non-multimodal model, the converted image411 triggers a clear error message instead of being ignored.412 """413 server.jinja = True414 server.start()415 416 # Small 1x1 red PNG image in base64 (same as vision tests)417 red_pixel_png = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg=="418 419 res = server.make_request("POST", "/v1/messages", data={420 "model": "test",421 "max_tokens": 100,422 "messages": [423 {"role": "user", "content": "What is in this image?"},424 {425 "role": "assistant",426 "content": [427 {428 "type": "tool_use",429 "id": "tool_1",430 "name": "read",431 "input": {"file": "test.png"}432 }433 ]434 },435 {436 "role": "user",437 "content": [438 {439 "type": "tool_result",440 "tool_use_id": "tool_1",441 "content": [442 {"type": "text", "text": "File: test.png"},443 {444 "type": "image",445 "source": {446 "type": "base64",447 "media_type": "image/png",448 "data": red_pixel_png449 }450 }451 ]452 }453 ]454 }455 ]456 })457 458 # Without the fix, image block would cause "unsupported content[].type"459 # With the fix, image is converted to image_url but tinyllama doesn't support images460 assert res.status_code == 500461 assert "image input is not supported" in res.body.get("error", {}).get("message", "").lower()462 463 464def test_anthropic_tool_result_error():465 """Test tool result with error flag"""466 server.jinja = True467 server.start()468 469 res = server.make_request("POST", "/v1/messages", data={470 "model": "test",471 "max_tokens": 100,472 "messages": [473 {"role": "user", "content": "Get the weather"},474 {475 "role": "assistant",476 "content": [477 {478 "type": "tool_use",479 "id": "test123",480 "name": "get_weather",481 "input": {"location": "InvalidCity"}482 }483 ]484 },485 {486 "role": "user",487 "content": [488 {489 "type": "tool_result",490 "tool_use_id": "test123",491 "is_error": True,492 "content": "City not found"493 }494 ]495 }496 ]497 })498 499 assert res.status_code == 200500 assert res.body["type"] == "message"501 502 503def test_anthropic_tool_streaming():504 """Test streaming with tool use"""505 server.jinja = True506 server.start()507 508 res = server.make_stream_request("POST", "/v1/messages", data={509 "model": "test",510 "max_tokens": 200,511 "stream": True,512 "tools": [{513 "name": "calculator",514 "description": "Calculate math",515 "input_schema": {516 "type": "object",517 "properties": {518 "expression": {"type": "string"}519 },520 "required": ["expression"]521 }522 }],523 "messages": [524 {"role": "user", "content": "Calculate 2+2"}525 ]526 })527 528 events = []529 for data in res:530 events.append(data)531 532 event_types = [e["type"] for e in events]533 534 # Should have basic events535 assert "message_start" in event_types536 assert "message_stop" in event_types537 538 # If tool was used, check for proper tool streaming539 if any(e.get("type") == "content_block_start" and540 e.get("content_block", {}).get("type") == "tool_use"541 for e in events):542 # Find tool use block start543 tool_starts = [e for e in events if544 e.get("type") == "content_block_start" and545 e.get("content_block", {}).get("type") == "tool_use"]546 547 assert len(tool_starts) > 0, "Should have tool_use content_block_start"548 549 # Check index is correct (should be 0 if no text, 1 if there's text)550 tool_start = tool_starts[0]551 assert "index" in tool_start552 assert tool_start["content_block"]["type"] == "tool_use"553 assert "name" in tool_start["content_block"]554 555 556# Vision/multimodal tests557 558def test_anthropic_vision_format_accepted():559 """Test that Anthropic vision format is accepted (format validation only)"""560 server.start()561 562 # Small 1x1 red PNG image in base64563 red_pixel_png = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg=="564 565 res = server.make_request("POST", "/v1/messages", data={566 "model": "test",567 "max_tokens": 10,568 "messages": [569 {570 "role": "user",571 "content": [572 {573 "type": "image",574 "source": {575 "type": "base64",576 "media_type": "image/png",577 "data": red_pixel_png578 }579 },580 {581 "type": "text",582 "text": "What is this?"583 }584 ]585 }586 ]587 })588 589 # Server accepts the format but tinyllama doesn't support images590 # So it should return 500 with clear error message about missing mmproj591 assert res.status_code == 500592 assert "image input is not supported" in res.body.get("error", {}).get("message", "").lower()593 594 595def test_anthropic_vision_base64_with_multimodal_model(vision_server):596 """Test vision with base64 image using Anthropic format with multimodal model"""597 global server598 server = vision_server599 server.start()600 601 # Get test image in base64 format602 image_base64 = get_test_image_base64()603 604 res = server.make_request("POST", "/v1/messages", data={605 "model": "test",606 "max_tokens": 10,607 "messages": [608 {609 "role": "user",610 "content": [611 {612 "type": "image",613 "source": {614 "type": "base64",615 "media_type": "image/png",616 "data": image_base64617 }618 },619 {620 "type": "text",621 "text": "What is this:\n"622 }623 ]624 }625 ]626 })627 628 assert res.status_code == 200, f"Expected 200, got {res.status_code}: {res.body}"629 assert res.body["type"] == "message"630 assert len(res.body["content"]) > 0631 assert res.body["content"][0]["type"] == "text"632 # The model should generate some response about the image633 assert len(res.body["content"][0]["text"]) > 0634 635 636# Parameter tests637 638def test_anthropic_stop_sequences():639 """Test stop_sequences parameter"""640 server.start()641 642 res = server.make_request("POST", "/v1/messages", data={643 "model": "test",644 "max_tokens": 100,645 "stop_sequences": ["\n", "END"],646 "messages": [647 {"role": "user", "content": "Count to 10"}648 ]649 })650 651 assert res.status_code == 200652 assert res.body["type"] == "message"653 654 655def test_anthropic_temperature():656 """Test temperature parameter"""657 server.start()658 659 res = server.make_request("POST", "/v1/messages", data={660 "model": "test",661 "max_tokens": 50,662 "temperature": 0.5,663 "messages": [664 {"role": "user", "content": "Hello"}665 ]666 })667 668 assert res.status_code == 200669 assert res.body["type"] == "message"670 671 672def test_anthropic_top_p():673 """Test top_p parameter"""674 server.start()675 676 res = server.make_request("POST", "/v1/messages", data={677 "model": "test",678 "max_tokens": 50,679 "top_p": 0.9,680 "messages": [681 {"role": "user", "content": "Hello"}682 ]683 })684 685 assert res.status_code == 200686 assert res.body["type"] == "message"687 688 689def test_anthropic_top_k():690 """Test top_k parameter (llama.cpp specific)"""691 server.start()692 693 res = server.make_request("POST", "/v1/messages", data={694 "model": "test",695 "max_tokens": 50,696 "top_k": 40,697 "messages": [698 {"role": "user", "content": "Hello"}699 ]700 })701 702 assert res.status_code == 200703 assert res.body["type"] == "message"704 705 706# Error handling tests707 708def test_anthropic_missing_messages():709 """Test error when messages are missing"""710 server.start()711 712 res = server.make_request("POST", "/v1/messages", data={713 "model": "test",714 "max_tokens": 50715 # missing "messages" field716 })717 718 # Should return an error (400 or 500)719 assert res.status_code >= 400720 721 722def test_anthropic_empty_messages():723 """Test permissive handling of empty messages array"""724 server.start()725 726 res = server.make_request("POST", "/v1/messages", data={727 "model": "test",728 "max_tokens": 50,729 "messages": []730 })731 732 # Server is permissive and accepts empty messages (provides defaults)733 # This matches the permissive validation design choice734 assert res.status_code == 200735 assert res.body["type"] == "message"736 737 738# Content block index tests739 740def test_anthropic_streaming_content_block_indices():741 """Test that content block indices are correct in streaming"""742 server.jinja = True743 server.start()744 745 # Request that might produce both text and tool use746 res = server.make_stream_request("POST", "/v1/messages", data={747 "model": "test",748 "max_tokens": 400,749 "stream": True,750 "tools": [{751 "name": "test_tool",752 "description": "A test tool",753 "input_schema": {754 "type": "object",755 "properties": {756 "param": {"type": "string"}757 },758 "required": ["param"]759 }760 }],761 "messages": [762 {"role": "user", "content": "Use the test tool"}763 ]764 })765 766 events = []767 for data in res:768 events.append(data)769 770 # Check content_block_start events have sequential indices771 block_starts = [e for e in events if e.get("type") == "content_block_start"]772 if len(block_starts) > 1:773 # If there are multiple blocks, indices should be sequential774 indices = [e["index"] for e in block_starts]775 expected_indices = list(range(len(block_starts)))776 assert indices == expected_indices, f"Expected indices {expected_indices}, got {indices}"777 778 # Check content_block_stop events match the starts779 block_stops = [e for e in events if e.get("type") == "content_block_stop"]780 start_indices = set(e["index"] for e in block_starts)781 stop_indices = set(e["index"] for e in block_stops)782 assert start_indices == stop_indices, "content_block_stop indices should match content_block_start indices"783 784 785# Extended features tests786 787def test_anthropic_thinking():788 """Test extended thinking parameter"""789 server.jinja = True790 server.start()791 792 res = server.make_request("POST", "/v1/messages", data={793 "model": "test",794 "max_tokens": 100,795 "thinking": {796 "type": "enabled",797 "budget_tokens": 50798 },799 "messages": [800 {"role": "user", "content": "What is 2+2?"}801 ]802 })803 804 assert res.status_code == 200805 assert res.body["type"] == "message"806 807 808def test_anthropic_metadata():809 """Test metadata parameter"""810 server.start()811 812 res = server.make_request("POST", "/v1/messages", data={813 "model": "test",814 "max_tokens": 50,815 "metadata": {816 "user_id": "test_user_123"817 },818 "messages": [819 {"role": "user", "content": "Hello"}820 ]821 })822 823 assert res.status_code == 200824 assert res.body["type"] == "message"825 826 827# Compatibility tests828 829def test_anthropic_vs_openai_different_response_format():830 """Verify Anthropic format is different from OpenAI format"""831 server.start()832 833 # Make OpenAI request834 openai_res = server.make_request("POST", "/v1/chat/completions", data={835 "model": "test",836 "max_tokens": 50,837 "messages": [838 {"role": "user", "content": "Hello"}839 ]840 })841 842 # Make Anthropic request843 anthropic_res = server.make_request("POST", "/v1/messages", data={844 "model": "test",845 "max_tokens": 50,846 "messages": [847 {"role": "user", "content": "Hello"}848 ]849 })850 851 assert openai_res.status_code == 200852 assert anthropic_res.status_code == 200853 854 # OpenAI has "object", Anthropic has "type"855 assert "object" in openai_res.body856 assert "type" in anthropic_res.body857 assert openai_res.body["object"] == "chat.completion"858 assert anthropic_res.body["type"] == "message"859 860 # OpenAI has "choices", Anthropic has "content"861 assert "choices" in openai_res.body862 assert "content" in anthropic_res.body863 864 # Different usage field names865 assert "prompt_tokens" in openai_res.body["usage"]866 assert "input_tokens" in anthropic_res.body["usage"]867 assert "completion_tokens" in openai_res.body["usage"]868 assert "output_tokens" in anthropic_res.body["usage"]869 870 871# Extended thinking tests with reasoning models872 873# The next two tests cover the input path (conversation history):874# Client sends thinking blocks -> convert_anthropic_to_oai -> reasoning_content -> template875 876def test_anthropic_thinking_history_in_count_tokens():877 """Test that interleaved thinking blocks in conversation history are not dropped during conversion."""878 global server879 server.jinja = True880 server.chat_template_file = '../../../models/templates/Qwen-Qwen3-0.6B.jinja'881 server.start()882 883 tool = {884 "name": "list_files",885 "description": "List files",886 "input_schema": {887 "type": "object",888 "properties": {"path": {"type": "string"}},889 "required": ["path"]890 }891 }892 893 messages_without_thinking = [894 {"role": "user", "content": "Fix the bug"},895 {896 "role": "assistant",897 "content": [898 {"type": "tool_use", "id": "call_1", "name": "list_files", "input": {"path": "."}}899 ]900 },901 {902 "role": "user",903 "content": [904 {"type": "tool_result", "tool_use_id": "call_1", "content": "main.py"}905 ]906 },907 ]908 909 messages_with_thinking = [910 {"role": "user", "content": "Fix the bug"},911 {912 "role": "assistant",913 "content": [914 {"type": "thinking", "thinking": "I should check the project structure first to understand the codebase layout."},915 {"type": "tool_use", "id": "call_1", "name": "list_files", "input": {"path": "."}}916 ]917 },918 {919 "role": "user",920 "content": [921 {"type": "tool_result", "tool_use_id": "call_1", "content": "main.py"}922 ]923 },924 ]925 926 res_without = server.make_request("POST", "/v1/messages/count_tokens", data={927 "model": "test",928 "messages": messages_without_thinking,929 "tools": [tool],930 })931 assert res_without.status_code == 200, f"Expected 200: {res_without.body}"932 933 res_with = server.make_request("POST", "/v1/messages/count_tokens", data={934 "model": "test",935 "messages": messages_with_thinking,936 "tools": [tool],937 })938 assert res_with.status_code == 200, f"Expected 200: {res_with.body}"939 940 # Thinking blocks should increase the token count941 assert res_with.body["input_tokens"] > res_without.body["input_tokens"], \942 f"Expected more tokens with thinking ({res_with.body['input_tokens']}) than without ({res_without.body['input_tokens']})"943 944 945def test_anthropic_thinking_history_in_template():946 """Test that reasoning_content from converted interleaved thinking blocks renders in the prompt."""947 global server948 server.jinja = True949 server.chat_template_file = '../../../models/templates/Qwen-Qwen3-0.6B.jinja'950 server.start()951 952 reasoning_1 = "I should check the project structure first."953 reasoning_2 = "Now I need to read the main file."954 955 res = server.make_request("POST", "/apply-template", data={956 "messages": [957 {"role": "user", "content": "Fix the bug in main.py"},958 {959 "role": "assistant",960 "content": "",961 "reasoning_content": reasoning_1,962 "tool_calls": [{963 "id": "call_1",964 "type": "function",965 "function": {"name": "list_files", "arguments": "{\"path\": \".\"}"}966 }]967 },968 {"role": "tool", "tool_call_id": "call_1", "content": "main.py\nutils.py"},969 {970 "role": "assistant",971 "content": "",972 "reasoning_content": reasoning_2,973 "tool_calls": [{974 "id": "call_2",975 "type": "function",976 "function": {"name": "read_file", "arguments": "{\"path\": \"main.py\"}"}977 }]978 },979 {"role": "tool", "tool_call_id": "call_2", "content": "print('hello')"},980 ],981 "tools": [{982 "type": "function",983 "function": {984 "name": "list_files",985 "description": "List files",986 "parameters": {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}987 }988 }, {989 "type": "function",990 "function": {991 "name": "read_file",992 "description": "Read a file",993 "parameters": {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}994 }995 }],996 })997 assert res.status_code == 200, f"Expected 200, got {res.status_code}: {res.body}"998 prompt = res.body["prompt"]999 1000 # Both reasoning_content values should be rendered in <think> tags1001 assert reasoning_1 in prompt, f"Expected first reasoning text in prompt: {prompt}"1002 assert reasoning_2 in prompt, f"Expected second reasoning text in prompt: {prompt}"1003 assert prompt.count("<think>") >= 2, f"Expected at least 2 <think> blocks in prompt: {prompt}"1004 1005 1006@pytest.mark.slow1007@pytest.mark.parametrize("stream", [False, True])1008def test_anthropic_thinking_with_reasoning_model(stream):1009 """Test that thinking content blocks are properly returned for reasoning models"""1010 global server1011 server = ServerProcess()1012 server.model_hf_repo = "bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF"1013 server.model_hf_file = "DeepSeek-R1-Distill-Qwen-7B-Q4_K_M.gguf"1014 server.reasoning_format = "deepseek"1015 server.jinja = True1016 server.n_ctx = 81921017 server.n_predict = 10241018 server.server_port = 80841019 server.start(timeout_seconds=600) # large model needs time to download1020 1021 if stream:1022 res = server.make_stream_request("POST", "/v1/messages", data={1023 "model": "test",1024 "max_tokens": 1024,1025 "thinking": {1026 "type": "enabled",1027 "budget_tokens": 5001028 },1029 "messages": [1030 {"role": "user", "content": "What is 2+2?"}1031 ],1032 "stream": True1033 })1034 1035 events = list(res)1036 1037 # should have thinking content block events1038 thinking_starts = [e for e in events if1039 e.get("type") == "content_block_start" and1040 e.get("content_block", {}).get("type") == "thinking"]1041 assert len(thinking_starts) > 0, "Should have thinking content_block_start event"1042 assert thinking_starts[0]["index"] == 0, "Thinking block should be at index 0"1043 1044 # should have thinking_delta events1045 thinking_deltas = [e for e in events if1046 e.get("type") == "content_block_delta" and1047 e.get("delta", {}).get("type") == "thinking_delta"]1048 assert len(thinking_deltas) > 0, "Should have thinking_delta events"1049 1050 # should have signature_delta event before thinking block closes (Anthropic API requirement)1051 signature_deltas = [e for e in events if1052 e.get("type") == "content_block_delta" and1053 e.get("delta", {}).get("type") == "signature_delta"]1054 assert len(signature_deltas) > 0, "Should have signature_delta event for thinking block"1055 1056 # should have text block after thinking1057 text_starts = [e for e in events if1058 e.get("type") == "content_block_start" and1059 e.get("content_block", {}).get("type") == "text"]1060 assert len(text_starts) > 0, "Should have text content_block_start event"1061 assert text_starts[0]["index"] == 1, "Text block should be at index 1 (after thinking)"1062 else:1063 res = server.make_request("POST", "/v1/messages", data={1064 "model": "test",1065 "max_tokens": 1024,1066 "thinking": {1067 "type": "enabled",1068 "budget_tokens": 5001069 },1070 "messages": [1071 {"role": "user", "content": "What is 2+2?"}1072 ]1073 })1074 1075 assert res.status_code == 2001076 assert res.body["type"] == "message"1077 1078 content = res.body["content"]1079 assert len(content) >= 2, "Should have at least thinking and text blocks"1080 1081 # first block should be thinking1082 thinking_blocks = [b for b in content if b.get("type") == "thinking"]1083 assert len(thinking_blocks) > 0, "Should have thinking content block"1084 assert "thinking" in thinking_blocks[0], "Thinking block should have 'thinking' field"1085 assert len(thinking_blocks[0]["thinking"]) > 0, "Thinking content should not be empty"1086 assert "signature" in thinking_blocks[0], "Thinking block should have 'signature' field (Anthropic API requirement)"1087 1088 # should also have text block1089 text_blocks = [b for b in content if b.get("type") == "text"]1090 assert len(text_blocks) > 0, "Should have text content block"1091 