Team Ai
Datasetpublic

echodict/llama.cpp

version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes479downloads
test_router.py208 linesDownload Raw Back to unit
1import pytest2from utils import *3 4server: ServerProcess5 6@pytest.fixture(autouse=True)7def create_server():8    global server9    server = ServerPreset.router()10 11 12def test_router_props():13    global server14    server.models_max = 215    server.no_models_autoload = True16    server.start()17    res = server.make_request("GET", "/props")18    assert res.status_code == 20019    assert res.body["role"] == "router"20    assert res.body["max_instances"] == 221    assert res.body["models_autoload"] is False22    assert res.body["build_info"].startswith("b")23 24 25@pytest.mark.parametrize(26    "model,success",27    [28        ("ggml-org/tinygemma3-GGUF:Q8_0", True),29        ("non-existent/model", False),30    ]31)32def test_router_chat_completion_stream(model: str, success: bool):33    global server34    server.start()35    content = ""36    ex: ServerError | None = None37    try:38        res = server.make_stream_request("POST", "/chat/completions", data={39            "model": model,40            "max_tokens": 16,41            "messages": [42                {"role": "user", "content": "hello"},43            ],44            "stream": True,45        })46        for data in res:47            if data["choices"]:48                choice = data["choices"][0]49                if choice["finish_reason"] in ["stop", "length"]:50                    assert "content" not in choice["delta"]51                else:52                    assert choice["finish_reason"] is None53                    content += choice["delta"]["content"] or ''54    except ServerError as e:55        ex = e56 57    if success:58        assert ex is None59        assert len(content) > 060    else:61        assert ex is not None62        assert content == ""63 64 65def _get_model_status(model_id: str) -> str:66    res = server.make_request("GET", "/models")67    assert res.status_code == 20068    for item in res.body.get("data", []):69        if item.get("id") == model_id or item.get("model") == model_id:70            return item["status"]["value"]71    raise AssertionError(f"Model {model_id} not found in /models response")72 73 74def _wait_for_model_status(model_id: str, desired: set[str], timeout: int = 60) -> str:75    deadline = time.time() + timeout76    last_status = None77    while time.time() < deadline:78        last_status = _get_model_status(model_id)79        if last_status in desired:80            return last_status81        time.sleep(1)82    raise AssertionError(83        f"Timed out waiting for {model_id} to reach {desired}, last status: {last_status}"84    )85 86 87def _load_model_and_wait(88    model_id: str, timeout: int = 60, headers: dict | None = None89) -> None:90    load_res = server.make_request(91        "POST", "/models/load", data={"model": model_id}, headers=headers92    )93    assert load_res.status_code == 20094    assert isinstance(load_res.body, dict)95    assert load_res.body.get("success") is True96    _wait_for_model_status(model_id, {"loaded"}, timeout=timeout)97 98 99def test_router_unload_model():100    global server101    server.start()102    model_id = "ggml-org/tinygemma3-GGUF:Q8_0"103 104    _load_model_and_wait(model_id)105 106    unload_res = server.make_request("POST", "/models/unload", data={"model": model_id})107    assert unload_res.status_code == 200108    assert unload_res.body.get("success") is True109    _wait_for_model_status(model_id, {"unloaded"})110 111 112def test_router_models_max_evicts_lru():113    global server114    server.models_max = 2115    server.start()116 117    candidate_models = [118        "ggml-org/tinygemma3-GGUF:Q8_0",119        "ggml-org/test-model-stories260K:F32",120        "ggml-org/test-model-stories260K-infill:F32",121    ]122 123    # Load only the first 2 models to fill the cache124    first, second, third = candidate_models[:3]125 126    _load_model_and_wait(first, timeout=120)127    _load_model_and_wait(second, timeout=120)128 129    # Verify both models are loaded130    assert _get_model_status(first) == "loaded"131    assert _get_model_status(second) == "loaded"132 133    # Load the third model - this should trigger LRU eviction of the first model134    _load_model_and_wait(third, timeout=120)135 136    # Verify eviction: third is loaded, first was evicted137    assert _get_model_status(third) == "loaded"138    assert _get_model_status(first) == "unloaded"139 140 141def test_router_no_models_autoload():142    global server143    server.no_models_autoload = True144    server.start()145    model_id = "ggml-org/tinygemma3-GGUF:Q8_0"146 147    res = server.make_request(148        "POST",149        "/v1/chat/completions",150        data={151            "model": model_id,152            "messages": [{"role": "user", "content": "hello"}],153            "max_tokens": 4,154        },155    )156    assert res.status_code == 400157    assert "error" in res.body158 159    _load_model_and_wait(model_id)160 161    success_res = server.make_request(162        "POST",163        "/v1/chat/completions",164        data={165            "model": model_id,166            "messages": [{"role": "user", "content": "hello"}],167            "max_tokens": 4,168        },169    )170    assert success_res.status_code == 200171    assert "error" not in success_res.body172 173 174def test_router_api_key_required():175    global server176    server.api_key = "sk-router-secret"177    server.start()178 179    model_id = "ggml-org/tinygemma3-GGUF:Q8_0"180    auth_headers = {"Authorization": f"Bearer {server.api_key}"}181 182    res = server.make_request(183        "POST",184        "/v1/chat/completions",185        data={186            "model": model_id,187            "messages": [{"role": "user", "content": "hello"}],188            "max_tokens": 4,189        },190    )191    assert res.status_code == 401192    assert res.body.get("error", {}).get("type") == "authentication_error"193 194    _load_model_and_wait(model_id, headers=auth_headers)195 196    authed = server.make_request(197        "POST",198        "/v1/chat/completions",199        headers=auth_headers,200        data={201            "model": model_id,202            "messages": [{"role": "user", "content": "hello"}],203            "max_tokens": 4,204        },205    )206    assert authed.status_code == 200207    assert "error" not in authed.body208