Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
test_stream.py154 linesDownload Raw Back to unit
1import json2import socket3import threading4import time5from urllib.parse import quote6import pytest7from utils import *8 9server: ServerProcess10 11# a model name with slashes exercises the query string routing of the stream routes: the id12# cannot travel as a path param because the decoded slash would split it before capture13MODEL = "ggml-org/tinygemma3-GGUF:Q8_0"14STREAM_ID = f"conv-stream-test::{MODEL}"15QS = "conv_id=" + quote(STREAM_ID, safe="")16 17 18@pytest.fixture(autouse=True)19def create_server():20    global server21    server = ServerPreset.router()22 23 24def test_stream_resume_and_stop_with_slashed_model_name():25    global server26    server.start()27 28    content = ""29    for data in server.make_stream_request("POST", "/chat/completions", data={30        "model": MODEL,31        "stream": True,32        "max_tokens": 16,33        "messages": [{"role": "user", "content": "hello"}],34    }, headers={"X-Conversation-Id": STREAM_ID}):35        if data["choices"]:36            content += data["choices"][0]["delta"].get("content") or ""37    assert len(content) > 038 39    # the finished session replays from the beginning through the router40    res = server.make_request("GET", f"/v1/stream?{QS}&from=0")41    assert res.status_code == 20042    assert "data: " in str(res.body)43 44    # the explicit stop reaches the owning child and evicts the session45    res = server.make_request("DELETE", f"/v1/stream?{QS}")46    assert res.status_code == 20447    res = server.make_request("GET", f"/v1/stream?{QS}&from=0")48    assert res.status_code == 40449 50 51def test_stream_stop_during_model_load():52    global server53    server.start()54 55    thread_error: list[ServerError] = []56    thread_done = threading.Event()57 58    def fire_post():59        try:60            for _ in server.make_stream_request("POST", "/chat/completions", data={61                "model": MODEL,62                "stream": True,63                "max_tokens": 512,64                "messages": [{"role": "user", "content": "Count from 1 to 1000."}],65            }, headers={"X-Conversation-Id": STREAM_ID}):66                pass67        except ServerError as e:68            thread_error.append(e)69        finally:70            thread_done.set()71 72    t = threading.Thread(target=fire_post)73    t.start()74 75    # catch the autoload window, tiny models load fast so poll aggressively76    saw_loading = False77    deadline = time.time() + 5.078    while time.time() < deadline and not thread_done.is_set():79        res = server.make_request("GET", "/models")80        status = next(m["status"]["value"] for m in res.body["data"] if m["id"] == MODEL)81        if status == "loading":82            saw_loading = True83            break84        time.sleep(0.002)85    if not saw_loading:86        t.join()87        pytest.skip("load window too short to be observed on this machine")  # ty: ignore[too-many-positional-arguments]88 89    # a stop during the load cancels the parked request instead of leaving an orphan90    res = server.make_request("DELETE", f"/v1/stream?{QS}")91    assert res.status_code == 20492    assert thread_done.wait(timeout=60)93    t.join()94    assert len(thread_error) == 195    assert thread_error[0].code == 40096    assert "cancelled" in json.dumps(thread_error[0].body)97    res = server.make_request("GET", f"/v1/stream?{QS}&from=0")98    assert res.status_code == 40499 100 101def test_stream_resumes_after_reload_during_model_load():102    global server103    server.start()104 105    # raw socket client so the connection can be dropped mid load like a page reload106    body = json.dumps({107        "model": MODEL,108        "stream": True,109        "max_tokens": 16,110        "messages": [{"role": "user", "content": "hello"}],111    })112    request = (113        f"POST /v1/chat/completions HTTP/1.1\r\n"114        f"Host: {server.server_host}:{server.server_port}\r\n"115        f"Content-Type: application/json\r\n"116        f"X-Conversation-Id: {STREAM_ID}\r\n"117        f"Content-Length: {len(body)}\r\n"118        f"Connection: close\r\n\r\n{body}"119    )120    sock = socket.create_connection((server.server_host, server.server_port))121    sock.sendall(request.encode())122 123    # drop the client while the model loads, poll aggressively to catch the window124    saw_loading = False125    saw_503 = False126    deadline = time.time() + 5.0127    while time.time() < deadline:128        res = server.make_request("GET", "/models")129        status = next(m["status"]["value"] for m in res.body["data"] if m["id"] == MODEL)130        if status == "loading":131            saw_loading = True132            break133        if status == "loaded":134            break135        time.sleep(0.002)136    sock.close()137    if not saw_loading:138        pytest.skip("load window too short to be observed on this machine")  # ty: ignore[too-many-positional-arguments]139 140    # while the model loads the resume route answers retry later, then the session appears,141    # receives the whole generation despite the dead client, and replays from the beginning142    deadline = time.time() + 60.0143    replay = None144    while time.time() < deadline:145        res = server.make_request("GET", f"/v1/stream?{QS}&from=0")146        if res.status_code == 503:147            saw_503 = True148        elif res.status_code == 200 and "data: " in str(res.body):149            replay = res150            break151        time.sleep(0.1)152    assert saw_503, "resume during the load did not answer 503"153    assert replay is not None, "session never became resumable after the client disconnect"154