Felipe97/llama-cpp-compiled
01.2k
1import json2import socket3import threading4import time5from urllib.parse import quote6import pytest7from utils import *8 9server: ServerProcess10 11# a model name with slashes exercises the query string routing of the stream routes: the id12# cannot travel as a path param because the decoded slash would split it before capture13MODEL = "ggml-org/tinygemma3-GGUF:Q8_0"14STREAM_ID = f"conv-stream-test::{MODEL}"15QS = "conv_id=" + quote(STREAM_ID, safe="")16 17 18@pytest.fixture(autouse=True)19def create_server():20 global server21 server = ServerPreset.router()22 23 24def test_stream_resume_and_stop_with_slashed_model_name():25 global server26 server.start()27 28 content = ""29 for data in server.make_stream_request("POST", "/chat/completions", data={30 "model": MODEL,31 "stream": True,32 "max_tokens": 16,33 "messages": [{"role": "user", "content": "hello"}],34 }, headers={"X-Conversation-Id": STREAM_ID}):35 if data["choices"]:36 content += data["choices"][0]["delta"].get("content") or ""37 assert len(content) > 038 39 # the finished session replays from the beginning through the router40 res = server.make_request("GET", f"/v1/stream?{QS}&from=0")41 assert res.status_code == 20042 assert "data: " in str(res.body)43 44 # the explicit stop reaches the owning child and evicts the session45 res = server.make_request("DELETE", f"/v1/stream?{QS}")46 assert res.status_code == 20447 res = server.make_request("GET", f"/v1/stream?{QS}&from=0")48 assert res.status_code == 40449 50 51def test_stream_stop_during_model_load():52 global server53 server.start()54 55 thread_error: list[ServerError] = []56 thread_done = threading.Event()57 58 def fire_post():59 try:60 for _ in server.make_stream_request("POST", "/chat/completions", data={61 "model": MODEL,62 "stream": True,63 "max_tokens": 512,64 "messages": [{"role": "user", "content": "Count from 1 to 1000."}],65 }, headers={"X-Conversation-Id": STREAM_ID}):66 pass67 except ServerError as e:68 thread_error.append(e)69 finally:70 thread_done.set()71 72 t = threading.Thread(target=fire_post)73 t.start()74 75 # catch the autoload window, tiny models load fast so poll aggressively76 saw_loading = False77 deadline = time.time() + 5.078 while time.time() < deadline and not thread_done.is_set():79 res = server.make_request("GET", "/models")80 status = next(m["status"]["value"] for m in res.body["data"] if m["id"] == MODEL)81 if status == "loading":82 saw_loading = True83 break84 time.sleep(0.002)85 if not saw_loading:86 t.join()87 pytest.skip("load window too short to be observed on this machine") # ty: ignore[too-many-positional-arguments]88 89 # a stop during the load cancels the parked request instead of leaving an orphan90 res = server.make_request("DELETE", f"/v1/stream?{QS}")91 assert res.status_code == 20492 assert thread_done.wait(timeout=60)93 t.join()94 assert len(thread_error) == 195 assert thread_error[0].code == 40096 assert "cancelled" in json.dumps(thread_error[0].body)97 res = server.make_request("GET", f"/v1/stream?{QS}&from=0")98 assert res.status_code == 40499 100 101def test_stream_resumes_after_reload_during_model_load():102 global server103 server.start()104 105 # raw socket client so the connection can be dropped mid load like a page reload106 body = json.dumps({107 "model": MODEL,108 "stream": True,109 "max_tokens": 16,110 "messages": [{"role": "user", "content": "hello"}],111 })112 request = (113 f"POST /v1/chat/completions HTTP/1.1\r\n"114 f"Host: {server.server_host}:{server.server_port}\r\n"115 f"Content-Type: application/json\r\n"116 f"X-Conversation-Id: {STREAM_ID}\r\n"117 f"Content-Length: {len(body)}\r\n"118 f"Connection: close\r\n\r\n{body}"119 )120 sock = socket.create_connection((server.server_host, server.server_port))121 sock.sendall(request.encode())122 123 # drop the client while the model loads, poll aggressively to catch the window124 saw_loading = False125 saw_503 = False126 deadline = time.time() + 5.0127 while time.time() < deadline:128 res = server.make_request("GET", "/models")129 status = next(m["status"]["value"] for m in res.body["data"] if m["id"] == MODEL)130 if status == "loading":131 saw_loading = True132 break133 if status == "loaded":134 break135 time.sleep(0.002)136 sock.close()137 if not saw_loading:138 pytest.skip("load window too short to be observed on this machine") # ty: ignore[too-many-positional-arguments]139 140 # while the model loads the resume route answers retry later, then the session appears,141 # receives the whole generation despite the dead client, and replays from the beginning142 deadline = time.time() + 60.0143 replay = None144 while time.time() < deadline:145 res = server.make_request("GET", f"/v1/stream?{QS}&from=0")146 if res.status_code == 503:147 saw_503 = True148 elif res.status_code == 200 and "data: " in str(res.body):149 replay = res150 break151 time.sleep(0.1)152 assert saw_503, "resume during the load did not answer 503"153 assert replay is not None, "session never became resumable after the client disconnect"154 