Felipe97/llama-cpp-compiled
01.2k
1#!/usr/bin/env python32import pytest3import base644import requests5 6from utils import *7 8server: ServerProcess9 10 11def get_test_image_base64() -> str:12 """Get a test image in base64 format"""13 # Use the same test image as test_vision_api.py14 IMG_URL = "https://huggingface.co/ggml-org/tinygemma3-GGUF/resolve/main/test/11_truck.png"15 response = requests.get(IMG_URL)16 response.raise_for_status()17 return base64.b64encode(response.content).decode("utf-8")18 19@pytest.fixture(autouse=True)20def create_server():21 global server22 server = ServerPreset.tinyllama2()23 server.model_alias = "tinyllama-2-anthropic"24 server.n_slots = 125 server.n_ctx = 819226 server.n_batch = 204827 28 29@pytest.fixture30def vision_server():31 """Separate fixture for vision tests that require multimodal support"""32 global server33 server = ServerPreset.tinygemma3()34 server.offline = False # Allow downloading the model35 server.model_alias = "tinygemma3-anthropic"36 server.n_slots = 137 return server38 39 40# Basic message tests41 42def test_anthropic_messages_basic():43 """Test basic Anthropic messages endpoint"""44 server.start()45 46 res = server.make_request("POST", "/v1/messages", data={47 "model": "test",48 "max_tokens": 50,49 "messages": [50 {"role": "user", "content": "Say hello"}51 ]52 })53 54 assert res.status_code == 200, f"Expected 200, got {res.status_code}"55 assert res.body["type"] == "message", f"Expected type 'message', got {res.body.get('type')}"56 assert res.body["role"] == "assistant", f"Expected role 'assistant', got {res.body.get('role')}"57 assert "content" in res.body, "Missing 'content' field"58 assert isinstance(res.body["content"], list), "Content should be an array"59 assert len(res.body["content"]) > 0, "Content array should not be empty"60 assert res.body["content"][0]["type"] == "text", "First content block should be text"61 assert "text" in res.body["content"][0], "Text content block missing 'text' field"62 assert res.body["stop_reason"] in ["end_turn", "max_tokens"], f"Invalid stop_reason: {res.body.get('stop_reason')}"63 assert "usage" in res.body, "Missing 'usage' field"64 assert "cache_read_input_tokens" in res.body["usage"], "Missing usage.cache_read_input_tokens"65 assert "input_tokens" in res.body["usage"], "Missing usage.input_tokens"66 assert "output_tokens" in res.body["usage"], "Missing usage.output_tokens"67 assert isinstance(res.body["usage"]["cache_read_input_tokens"], int), "cache_read_input_tokens should be integer"68 assert isinstance(res.body["usage"]["input_tokens"], int), "input_tokens should be integer"69 assert isinstance(res.body["usage"]["output_tokens"], int), "output_tokens should be integer"70 assert res.body["usage"]["output_tokens"] > 0, "Should have generated some tokens"71 # Anthropic API should NOT include timings72 assert "timings" not in res.body, "Anthropic API should not include timings field"73 74 75def test_anthropic_messages_with_system():76 """Test messages with system prompt"""77 server.start()78 79 res = server.make_request("POST", "/v1/messages", data={80 "model": "test",81 "max_tokens": 50,82 "system": "You are a helpful assistant.",83 "messages": [84 {"role": "user", "content": "Hello"}85 ]86 })87 88 assert res.status_code == 20089 assert res.body["type"] == "message"90 assert len(res.body["content"]) > 091 92 93def test_anthropic_messages_multipart_content():94 """Test messages with multipart content blocks"""95 server.start()96 97 res = server.make_request("POST", "/v1/messages", data={98 "model": "test",99 "max_tokens": 50,100 "messages": [101 {102 "role": "user",103 "content": [104 {"type": "text", "text": "What is"},105 {"type": "text", "text": " the answer?"}106 ]107 }108 ]109 })110 111 assert res.status_code == 200112 assert res.body["type"] == "message"113 114 115def test_anthropic_messages_conversation():116 """Test multi-turn conversation"""117 server.start()118 119 res = server.make_request("POST", "/v1/messages", data={120 "model": "test",121 "max_tokens": 50,122 "messages": [123 {"role": "user", "content": "Hello"},124 {"role": "assistant", "content": "Hi there!"},125 {"role": "user", "content": "How are you?"}126 ]127 })128 129 assert res.status_code == 200130 assert res.body["type"] == "message"131 132 133# Streaming tests134 135def test_anthropic_messages_streaming():136 """Test streaming messages"""137 server.start()138 139 res = server.make_stream_request("POST", "/v1/messages", data={140 "model": "test",141 "max_tokens": 30,142 "messages": [143 {"role": "user", "content": "Say hello"}144 ],145 "stream": True146 })147 148 events = []149 for data in res:150 # Each event should have type and other fields151 assert "type" in data, f"Missing 'type' in event: {data}"152 events.append(data)153 154 # Verify event sequence155 event_types = [e["type"] for e in events]156 assert "message_start" in event_types, "Missing message_start event"157 assert "content_block_start" in event_types, "Missing content_block_start event"158 assert "content_block_delta" in event_types, "Missing content_block_delta event"159 assert "content_block_stop" in event_types, "Missing content_block_stop event"160 assert "message_delta" in event_types, "Missing message_delta event"161 assert "message_stop" in event_types, "Missing message_stop event"162 163 # Check message_start structure164 message_start = next(e for e in events if e["type"] == "message_start")165 assert "message" in message_start, "message_start missing 'message' field"166 assert message_start["message"]["type"] == "message"167 assert message_start["message"]["role"] == "assistant"168 assert message_start["message"]["content"] == []169 assert "usage" in message_start["message"]170 assert message_start["message"]["usage"]["input_tokens"] > 0171 172 # Check content_block_start173 block_start = next(e for e in events if e["type"] == "content_block_start")174 assert "index" in block_start, "content_block_start missing 'index'"175 assert block_start["index"] == 0, "First content block should be at index 0"176 assert "content_block" in block_start177 assert block_start["content_block"]["type"] == "text"178 179 # Check content_block_delta180 deltas = [e for e in events if e["type"] == "content_block_delta"]181 assert len(deltas) > 0, "Should have at least one content_block_delta"182 for delta in deltas:183 assert "index" in delta184 assert "delta" in delta185 assert delta["delta"]["type"] == "text_delta"186 assert "text" in delta["delta"]187 188 # Check content_block_stop189 block_stop = next(e for e in events if e["type"] == "content_block_stop")190 assert "index" in block_stop191 assert block_stop["index"] == 0192 193 # Check message_delta194 message_delta = next(e for e in events if e["type"] == "message_delta")195 assert "delta" in message_delta196 assert "stop_reason" in message_delta["delta"]197 assert message_delta["delta"]["stop_reason"] in ["end_turn", "max_tokens"]198 assert "usage" in message_delta199 assert message_delta["usage"]["output_tokens"] > 0200 201 # Check message_stop202 message_stop = next(e for e in events if e["type"] == "message_stop")203 # message_stop should NOT have timings for Anthropic API204 assert "timings" not in message_stop, "Anthropic streaming should not include timings"205 206 207# Token counting tests208 209def test_anthropic_count_tokens():210 """Test token counting endpoint"""211 server.start()212 213 res = server.make_request("POST", "/v1/messages/count_tokens", data={214 "model": "test",215 "messages": [216 {"role": "user", "content": "Hello world"}217 ]218 })219 220 assert res.status_code == 200221 assert "input_tokens" in res.body222 assert isinstance(res.body["input_tokens"], int)223 assert res.body["input_tokens"] > 0224 # Should only have input_tokens, no other fields225 assert "output_tokens" not in res.body226 227 228def test_anthropic_count_tokens_with_system():229 """Test token counting with system prompt"""230 server.start()231 232 res = server.make_request("POST", "/v1/messages/count_tokens", data={233 "model": "test",234 "system": "You are a helpful assistant.",235 "messages": [236 {"role": "user", "content": "Hello"}237 ]238 })239 240 assert res.status_code == 200241 assert res.body["input_tokens"] > 0242 243 244def test_anthropic_count_tokens_no_max_tokens():245 """Test that count_tokens doesn't require max_tokens"""246 server.start()247 248 # max_tokens is NOT required for count_tokens249 res = server.make_request("POST", "/v1/messages/count_tokens", data={250 "model": "test",251 "messages": [252 {"role": "user", "content": "Hello"}253 ]254 })255 256 assert res.status_code == 200257 assert "input_tokens" in res.body258 259 260# Tool use tests261 262def test_anthropic_tool_use_basic():263 """Test basic tool use"""264 server.jinja = True265 server.start()266 267 res = server.make_request("POST", "/v1/messages", data={268 "model": "test",269 "max_tokens": 200,270 "tools": [{271 "name": "get_weather",272 "description": "Get the current weather in a location",273 "input_schema": {274 "type": "object",275 "properties": {276 "location": {277 "type": "string",278 "description": "City name"279 }280 },281 "required": ["location"]282 }283 }],284 "messages": [285 {"role": "user", "content": "What's the weather in Paris?"}286 ]287 })288 289 assert res.status_code == 200290 assert res.body["type"] == "message"291 assert len(res.body["content"]) > 0292 293 # Check if model used the tool (it might not always, depending on the model)294 content_types = [block.get("type") for block in res.body["content"]]295 296 if "tool_use" in content_types:297 # Model used the tool298 assert res.body["stop_reason"] == "tool_use"299 300 # Find the tool_use block301 tool_block = next(b for b in res.body["content"] if b.get("type") == "tool_use")302 assert "id" in tool_block303 assert "name" in tool_block304 assert tool_block["name"] == "get_weather"305 assert "input" in tool_block306 assert isinstance(tool_block["input"], dict)307 308 309def test_anthropic_tool_result():310 """Test sending tool results back311 312 This test verifies that tool_result blocks are properly converted to313 role="tool" messages internally. Without proper conversion, this would314 fail with a 500 error: "unsupported content[].type" because tool_result315 blocks would remain in the user message content array.316 """317 server.jinja = True318 server.start()319 320 res = server.make_request("POST", "/v1/messages", data={321 "model": "test",322 "max_tokens": 100,323 "messages": [324 {"role": "user", "content": "What's the weather?"},325 {326 "role": "assistant",327 "content": [328 {329 "type": "tool_use",330 "id": "test123",331 "name": "get_weather",332 "input": {"location": "Paris"}333 }334 ]335 },336 {337 "role": "user",338 "content": [339 {340 "type": "tool_result",341 "tool_use_id": "test123",342 "content": "The weather is sunny, 25ยฐC"343 }344 ]345 }346 ]347 })348 349 # This would be 500 with the old bug where tool_result blocks weren't converted350 assert res.status_code == 200351 assert res.body["type"] == "message"352 # Model should respond to the tool result353 assert len(res.body["content"]) > 0354 assert res.body["content"][0]["type"] == "text"355 356 357def test_anthropic_tool_result_with_text():358 """Test tool result mixed with text content359 360 This tests the edge case where a user message contains both text and361 tool_result blocks. The server must properly split these into separate362 messages: a user message with text, followed by tool messages.363 Without proper handling, this would fail with 500: "unsupported content[].type"364 """365 server.jinja = True366 server.start()367 368 res = server.make_request("POST", "/v1/messages", data={369 "model": "test",370 "max_tokens": 100,371 "messages": [372 {"role": "user", "content": "What's the weather?"},373 {374 "role": "assistant",375 "content": [376 {377 "type": "tool_use",378 "id": "tool_1",379 "name": "get_weather",380 "input": {"location": "Paris"}381 }382 ]383 },384 {385 "role": "user",386 "content": [387 {"type": "text", "text": "Here are the results:"},388 {389 "type": "tool_result",390 "tool_use_id": "tool_1",391 "content": "Sunny, 25ยฐC"392 }393 ]394 }395 ]396 })397 398 assert res.status_code == 200399 assert res.body["type"] == "message"400 assert len(res.body["content"]) > 0401 402 403def test_anthropic_tool_result_with_image():404 """Test tool result containing mixed text and image blocks405 406 Verifies that image blocks inside Anthropic tool_result content are407 properly converted to OpenAI image_url format rather than being408 silently dropped. With a non-multimodal model, the converted image409 triggers a clear error message instead of being ignored.410 """411 server.jinja = True412 server.start()413 414 # Small 1x1 red PNG image in base64 (same as vision tests)415 red_pixel_png = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg=="416 417 res = server.make_request("POST", "/v1/messages", data={418 "model": "test",419 "max_tokens": 100,420 "messages": [421 {"role": "user", "content": "What is in this image?"},422 {423 "role": "assistant",424 "content": [425 {426 "type": "tool_use",427 "id": "tool_1",428 "name": "read",429 "input": {"file": "test.png"}430 }431 ]432 },433 {434 "role": "user",435 "content": [436 {437 "type": "tool_result",438 "tool_use_id": "tool_1",439 "content": [440 {"type": "text", "text": "File: test.png"},441 {442 "type": "image",443 "source": {444 "type": "base64",445 "media_type": "image/png",446 "data": red_pixel_png447 }448 }449 ]450 }451 ]452 }453 ]454 })455 456 # Without the fix, image block would cause "unsupported content[].type"457 # With the fix, image is converted to image_url but tinyllama doesn't support images458 assert res.status_code == 500459 assert "image input is not supported" in res.body.get("error", {}).get("message", "").lower()460 461 462def test_anthropic_tool_result_error():463 """Test tool result with error flag"""464 server.jinja = True465 server.start()466 467 res = server.make_request("POST", "/v1/messages", data={468 "model": "test",469 "max_tokens": 100,470 "messages": [471 {"role": "user", "content": "Get the weather"},472 {473 "role": "assistant",474 "content": [475 {476 "type": "tool_use",477 "id": "test123",478 "name": "get_weather",479 "input": {"location": "InvalidCity"}480 }481 ]482 },483 {484 "role": "user",485 "content": [486 {487 "type": "tool_result",488 "tool_use_id": "test123",489 "is_error": True,490 "content": "City not found"491 }492 ]493 }494 ]495 })496 497 assert res.status_code == 200498 assert res.body["type"] == "message"499 500 501def test_anthropic_tool_streaming():502 """Test streaming with tool use"""503 server.jinja = True504 server.start()505 506 res = server.make_stream_request("POST", "/v1/messages", data={507 "model": "test",508 "max_tokens": 200,509 "stream": True,510 "tools": [{511 "name": "calculator",512 "description": "Calculate math",513 "input_schema": {514 "type": "object",515 "properties": {516 "expression": {"type": "string"}517 },518 "required": ["expression"]519 }520 }],521 "messages": [522 {"role": "user", "content": "Calculate 2+2"}523 ]524 })525 526 events = []527 for data in res:528 events.append(data)529 530 event_types = [e["type"] for e in events]531 532 # Should have basic events533 assert "message_start" in event_types534 assert "message_stop" in event_types535 536 # If tool was used, check for proper tool streaming537 if any(e.get("type") == "content_block_start" and538 e.get("content_block", {}).get("type") == "tool_use"539 for e in events):540 # Find tool use block start541 tool_starts = [e for e in events if542 e.get("type") == "content_block_start" and543 e.get("content_block", {}).get("type") == "tool_use"]544 545 assert len(tool_starts) > 0, "Should have tool_use content_block_start"546 547 # Check index is correct (should be 0 if no text, 1 if there's text)548 tool_start = tool_starts[0]549 assert "index" in tool_start550 assert tool_start["content_block"]["type"] == "tool_use"551 assert "name" in tool_start["content_block"]552 553 554# Vision/multimodal tests555 556def test_anthropic_vision_format_accepted():557 """Test that Anthropic vision format is accepted (format validation only)"""558 server.start()559 560 # Small 1x1 red PNG image in base64561 red_pixel_png = "iVBORw0KGgoAAAANSUhEUgAAAAEAAAABCAYAAAAfFcSJAAAADUlEQVR42mP8z8DwHwAFBQIAX8jx0gAAAABJRU5ErkJggg=="562 563 res = server.make_request("POST", "/v1/messages", data={564 "model": "test",565 "max_tokens": 10,566 "messages": [567 {568 "role": "user",569 "content": [570 {571 "type": "image",572 "source": {573 "type": "base64",574 "media_type": "image/png",575 "data": red_pixel_png576 }577 },578 {579 "type": "text",580 "text": "What is this?"581 }582 ]583 }584 ]585 })586 587 # Server accepts the format but tinyllama doesn't support images588 # So it should return 500 with clear error message about missing mmproj589 assert res.status_code == 500590 assert "image input is not supported" in res.body.get("error", {}).get("message", "").lower()591 592 593def test_anthropic_vision_base64_with_multimodal_model(vision_server):594 """Test vision with base64 image using Anthropic format with multimodal model"""595 global server596 server = vision_server597 server.start()598 599 # Get test image in base64 format600 image_base64 = get_test_image_base64()601 602 res = server.make_request("POST", "/v1/messages", data={603 "model": "test",604 "max_tokens": 10,605 "messages": [606 {607 "role": "user",608 "content": [609 {610 "type": "image",611 "source": {612 "type": "base64",613 "media_type": "image/png",614 "data": image_base64615 }616 },617 {618 "type": "text",619 "text": "What is this:\n"620 }621 ]622 }623 ]624 })625 626 assert res.status_code == 200, f"Expected 200, got {res.status_code}: {res.body}"627 assert res.body["type"] == "message"628 assert len(res.body["content"]) > 0629 assert res.body["content"][0]["type"] == "text"630 # The model should generate some response about the image631 assert len(res.body["content"][0]["text"]) > 0632 633 634# Parameter tests635 636def test_anthropic_stop_sequences():637 """Test stop_sequences parameter"""638 server.start()639 640 res = server.make_request("POST", "/v1/messages", data={641 "model": "test",642 "max_tokens": 100,643 "stop_sequences": ["\n", "END"],644 "messages": [645 {"role": "user", "content": "Count to 10"}646 ]647 })648 649 assert res.status_code == 200650 assert res.body["type"] == "message"651 652 653def test_anthropic_temperature():654 """Test temperature parameter"""655 server.start()656 657 res = server.make_request("POST", "/v1/messages", data={658 "model": "test",659 "max_tokens": 50,660 "temperature": 0.5,661 "messages": [662 {"role": "user", "content": "Hello"}663 ]664 })665 666 assert res.status_code == 200667 assert res.body["type"] == "message"668 669 670def test_anthropic_top_p():671 """Test top_p parameter"""672 server.start()673 674 res = server.make_request("POST", "/v1/messages", data={675 "model": "test",676 "max_tokens": 50,677 "top_p": 0.9,678 "messages": [679 {"role": "user", "content": "Hello"}680 ]681 })682 683 assert res.status_code == 200684 assert res.body["type"] == "message"685 686 687def test_anthropic_top_k():688 """Test top_k parameter (llama.cpp specific)"""689 server.start()690 691 res = server.make_request("POST", "/v1/messages", data={692 "model": "test",693 "max_tokens": 50,694 "top_k": 40,695 "messages": [696 {"role": "user", "content": "Hello"}697 ]698 })699 700 assert res.status_code == 200701 assert res.body["type"] == "message"702 703 704# Error handling tests705 706def test_anthropic_missing_messages():707 """Test error when messages are missing"""708 server.start()709 710 res = server.make_request("POST", "/v1/messages", data={711 "model": "test",712 "max_tokens": 50713 # missing "messages" field714 })715 716 # Should return an error (400 or 500)717 assert res.status_code >= 400718 719 720def test_anthropic_empty_messages():721 """Test permissive handling of empty messages array"""722 server.start()723 724 res = server.make_request("POST", "/v1/messages", data={725 "model": "test",726 "max_tokens": 50,727 "messages": []728 })729 730 # Server is permissive and accepts empty messages (provides defaults)731 # This matches the permissive validation design choice732 assert res.status_code == 200733 assert res.body["type"] == "message"734 735 736# Content block index tests737 738def test_anthropic_streaming_content_block_indices():739 """Test that content block indices are correct in streaming"""740 server.jinja = True741 server.start()742 743 # Request that might produce both text and tool use744 res = server.make_stream_request("POST", "/v1/messages", data={745 "model": "test",746 "max_tokens": 400,747 "stream": True,748 "tools": [{749 "name": "test_tool",750 "description": "A test tool",751 "input_schema": {752 "type": "object",753 "properties": {754 "param": {"type": "string"}755 },756 "required": ["param"]757 }758 }],759 "messages": [760 {"role": "user", "content": "Use the test tool"}761 ]762 })763 764 events = []765 for data in res:766 events.append(data)767 768 # Check content_block_start events have sequential indices769 block_starts = [e for e in events if e.get("type") == "content_block_start"]770 if len(block_starts) > 1:771 # If there are multiple blocks, indices should be sequential772 indices = [e["index"] for e in block_starts]773 expected_indices = list(range(len(block_starts)))774 assert indices == expected_indices, f"Expected indices {expected_indices}, got {indices}"775 776 # Check content_block_stop events match the starts777 block_stops = [e for e in events if e.get("type") == "content_block_stop"]778 start_indices = set(e["index"] for e in block_starts)779 stop_indices = set(e["index"] for e in block_stops)780 assert start_indices == stop_indices, "content_block_stop indices should match content_block_start indices"781 782 783# Extended features tests784 785def test_anthropic_thinking():786 """Test extended thinking parameter"""787 server.jinja = True788 server.start()789 790 res = server.make_request("POST", "/v1/messages", data={791 "model": "test",792 "max_tokens": 100,793 "thinking": {794 "type": "enabled",795 "budget_tokens": 50796 },797 "messages": [798 {"role": "user", "content": "What is 2+2?"}799 ]800 })801 802 assert res.status_code == 200803 assert res.body["type"] == "message"804 805 806def test_anthropic_metadata():807 """Test metadata parameter"""808 server.start()809 810 res = server.make_request("POST", "/v1/messages", data={811 "model": "test",812 "max_tokens": 50,813 "metadata": {814 "user_id": "test_user_123"815 },816 "messages": [817 {"role": "user", "content": "Hello"}818 ]819 })820 821 assert res.status_code == 200822 assert res.body["type"] == "message"823 824 825# Compatibility tests826 827def test_anthropic_vs_openai_different_response_format():828 """Verify Anthropic format is different from OpenAI format"""829 server.start()830 831 # Make OpenAI request832 openai_res = server.make_request("POST", "/v1/chat/completions", data={833 "model": "test",834 "max_tokens": 50,835 "messages": [836 {"role": "user", "content": "Hello"}837 ]838 })839 840 # Make Anthropic request841 anthropic_res = server.make_request("POST", "/v1/messages", data={842 "model": "test",843 "max_tokens": 50,844 "messages": [845 {"role": "user", "content": "Hello"}846 ]847 })848 849 assert openai_res.status_code == 200850 assert anthropic_res.status_code == 200851 852 # OpenAI has "object", Anthropic has "type"853 assert "object" in openai_res.body854 assert "type" in anthropic_res.body855 assert openai_res.body["object"] == "chat.completion"856 assert anthropic_res.body["type"] == "message"857 858 # OpenAI has "choices", Anthropic has "content"859 assert "choices" in openai_res.body860 assert "content" in anthropic_res.body861 862 # Different usage field names863 assert "prompt_tokens" in openai_res.body["usage"]864 assert "input_tokens" in anthropic_res.body["usage"]865 assert "completion_tokens" in openai_res.body["usage"]866 assert "output_tokens" in anthropic_res.body["usage"]867 868 869# Extended thinking tests with reasoning models870 871# The next two tests cover the input path (conversation history):872# Client sends thinking blocks -> convert_anthropic_to_oai -> reasoning_content -> template873 874def test_anthropic_thinking_history_in_count_tokens():875 """Test that interleaved thinking blocks in conversation history are not dropped during conversion."""876 global server877 server.jinja = True878 server.chat_template_file = '../../../models/templates/Qwen-Qwen3-0.6B.jinja'879 server.start()880 881 tool = {882 "name": "list_files",883 "description": "List files",884 "input_schema": {885 "type": "object",886 "properties": {"path": {"type": "string"}},887 "required": ["path"]888 }889 }890 891 messages_without_thinking = [892 {"role": "user", "content": "Fix the bug"},893 {894 "role": "assistant",895 "content": [896 {"type": "tool_use", "id": "call_1", "name": "list_files", "input": {"path": "."}}897 ]898 },899 {900 "role": "user",901 "content": [902 {"type": "tool_result", "tool_use_id": "call_1", "content": "main.py"}903 ]904 },905 ]906 907 messages_with_thinking = [908 {"role": "user", "content": "Fix the bug"},909 {910 "role": "assistant",911 "content": [912 {"type": "thinking", "thinking": "I should check the project structure first to understand the codebase layout."},913 {"type": "tool_use", "id": "call_1", "name": "list_files", "input": {"path": "."}}914 ]915 },916 {917 "role": "user",918 "content": [919 {"type": "tool_result", "tool_use_id": "call_1", "content": "main.py"}920 ]921 },922 ]923 924 res_without = server.make_request("POST", "/v1/messages/count_tokens", data={925 "model": "test",926 "messages": messages_without_thinking,927 "tools": [tool],928 })929 assert res_without.status_code == 200, f"Expected 200: {res_without.body}"930 931 res_with = server.make_request("POST", "/v1/messages/count_tokens", data={932 "model": "test",933 "messages": messages_with_thinking,934 "tools": [tool],935 })936 assert res_with.status_code == 200, f"Expected 200: {res_with.body}"937 938 # Thinking blocks should increase the token count939 assert res_with.body["input_tokens"] > res_without.body["input_tokens"], \940 f"Expected more tokens with thinking ({res_with.body['input_tokens']}) than without ({res_without.body['input_tokens']})"941 942 943def test_anthropic_thinking_history_in_template():944 """Test that reasoning_content from converted interleaved thinking blocks renders in the prompt."""945 global server946 server.jinja = True947 server.chat_template_file = '../../../models/templates/Qwen-Qwen3-0.6B.jinja'948 server.start()949 950 reasoning_1 = "I should check the project structure first."951 reasoning_2 = "Now I need to read the main file."952 953 res = server.make_request("POST", "/apply-template", data={954 "messages": [955 {"role": "user", "content": "Fix the bug in main.py"},956 {957 "role": "assistant",958 "content": "",959 "reasoning_content": reasoning_1,960 "tool_calls": [{961 "id": "call_1",962 "type": "function",963 "function": {"name": "list_files", "arguments": "{\"path\": \".\"}"}964 }]965 },966 {"role": "tool", "tool_call_id": "call_1", "content": "main.py\nutils.py"},967 {968 "role": "assistant",969 "content": "",970 "reasoning_content": reasoning_2,971 "tool_calls": [{972 "id": "call_2",973 "type": "function",974 "function": {"name": "read_file", "arguments": "{\"path\": \"main.py\"}"}975 }]976 },977 {"role": "tool", "tool_call_id": "call_2", "content": "print('hello')"},978 ],979 "tools": [{980 "type": "function",981 "function": {982 "name": "list_files",983 "description": "List files",984 "parameters": {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}985 }986 }, {987 "type": "function",988 "function": {989 "name": "read_file",990 "description": "Read a file",991 "parameters": {"type": "object", "properties": {"path": {"type": "string"}}, "required": ["path"]}992 }993 }],994 })995 assert res.status_code == 200, f"Expected 200, got {res.status_code}: {res.body}"996 prompt = res.body["prompt"]997 998 # Both reasoning_content values should be rendered in <think> tags999 assert reasoning_1 in prompt, f"Expected first reasoning text in prompt: {prompt}"1000 assert reasoning_2 in prompt, f"Expected second reasoning text in prompt: {prompt}"1001 assert prompt.count("<think>") >= 2, f"Expected at least 2 <think> blocks in prompt: {prompt}"1002 1003 1004@pytest.mark.slow1005@pytest.mark.parametrize("stream", [False, True])1006def test_anthropic_thinking_with_reasoning_model(stream):1007 """Test that thinking content blocks are properly returned for reasoning models"""1008 global server1009 server = ServerProcess()1010 server.model_hf_repo = "bartowski/DeepSeek-R1-Distill-Qwen-7B-GGUF"1011 server.model_hf_file = "DeepSeek-R1-Distill-Qwen-7B-Q4_K_M.gguf"1012 server.reasoning_format = "deepseek"1013 server.jinja = True1014 server.n_ctx = 81921015 server.n_predict = 10241016 server.start(timeout_seconds=600) # large model needs time to download1017 1018 if stream:1019 res = server.make_stream_request("POST", "/v1/messages", data={1020 "model": "test",1021 "max_tokens": 1024,1022 "thinking": {1023 "type": "enabled",1024 "budget_tokens": 5001025 },1026 "messages": [1027 {"role": "user", "content": "What is 2+2?"}1028 ],1029 "stream": True1030 })1031 1032 events = list(res)1033 1034 # should have thinking content block events1035 thinking_starts = [e for e in events if1036 e.get("type") == "content_block_start" and1037 e.get("content_block", {}).get("type") == "thinking"]1038 assert len(thinking_starts) > 0, "Should have thinking content_block_start event"1039 assert thinking_starts[0]["index"] == 0, "Thinking block should be at index 0"1040 1041 # should have thinking_delta events1042 thinking_deltas = [e for e in events if1043 e.get("type") == "content_block_delta" and1044 e.get("delta", {}).get("type") == "thinking_delta"]1045 assert len(thinking_deltas) > 0, "Should have thinking_delta events"1046 1047 # should have signature_delta event before thinking block closes (Anthropic API requirement)1048 signature_deltas = [e for e in events if1049 e.get("type") == "content_block_delta" and1050 e.get("delta", {}).get("type") == "signature_delta"]1051 assert len(signature_deltas) > 0, "Should have signature_delta event for thinking block"1052 1053 # should have text block after thinking1054 text_starts = [e for e in events if1055 e.get("type") == "content_block_start" and1056 e.get("content_block", {}).get("type") == "text"]1057 assert len(text_starts) > 0, "Should have text content_block_start event"1058 assert text_starts[0]["index"] == 1, "Text block should be at index 1 (after thinking)"1059 else:1060 res = server.make_request("POST", "/v1/messages", data={1061 "model": "test",1062 "max_tokens": 1024,1063 "thinking": {1064 "type": "enabled",1065 "budget_tokens": 5001066 },1067 "messages": [1068 {"role": "user", "content": "What is 2+2?"}1069 ]1070 })1071 1072 assert res.status_code == 2001073 assert res.body["type"] == "message"1074 1075 content = res.body["content"]1076 assert len(content) >= 2, "Should have at least thinking and text blocks"1077 1078 # first block should be thinking1079 thinking_blocks = [b for b in content if b.get("type") == "thinking"]1080 assert len(thinking_blocks) > 0, "Should have thinking content block"1081 assert "thinking" in thinking_blocks[0], "Thinking block should have 'thinking' field"1082 assert len(thinking_blocks[0]["thinking"]) > 0, "Thinking content should not be empty"1083 assert "signature" in thinking_blocks[0], "Thinking block should have 'signature' field (Anthropic API requirement)"1084 1085 # should also have text block1086 text_blocks = [b for b in content if b.get("type") == "text"]1087 assert len(text_blocks) > 0, "Should have text content block"1088 