echodict/llama.cpp
version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786
0479
1import pytest2from utils import *3 4server: ServerProcess5 6@pytest.fixture(autouse=True)7def create_server():8 global server9 server = ServerPreset.router()10 11 12def test_router_props():13 global server14 server.models_max = 215 server.no_models_autoload = True16 server.start()17 res = server.make_request("GET", "/props")18 assert res.status_code == 20019 assert res.body["role"] == "router"20 assert res.body["max_instances"] == 221 assert res.body["models_autoload"] is False22 assert res.body["build_info"].startswith("b")23 24 25@pytest.mark.parametrize(26 "model,success",27 [28 ("ggml-org/tinygemma3-GGUF:Q8_0", True),29 ("non-existent/model", False),30 ]31)32def test_router_chat_completion_stream(model: str, success: bool):33 global server34 server.start()35 content = ""36 ex: ServerError | None = None37 try:38 res = server.make_stream_request("POST", "/chat/completions", data={39 "model": model,40 "max_tokens": 16,41 "messages": [42 {"role": "user", "content": "hello"},43 ],44 "stream": True,45 })46 for data in res:47 if data["choices"]:48 choice = data["choices"][0]49 if choice["finish_reason"] in ["stop", "length"]:50 assert "content" not in choice["delta"]51 else:52 assert choice["finish_reason"] is None53 content += choice["delta"]["content"] or ''54 except ServerError as e:55 ex = e56 57 if success:58 assert ex is None59 assert len(content) > 060 else:61 assert ex is not None62 assert content == ""63 64 65def _get_model_status(model_id: str) -> str:66 res = server.make_request("GET", "/models")67 assert res.status_code == 20068 for item in res.body.get("data", []):69 if item.get("id") == model_id or item.get("model") == model_id:70 return item["status"]["value"]71 raise AssertionError(f"Model {model_id} not found in /models response")72 73 74def _wait_for_model_status(model_id: str, desired: set[str], timeout: int = 60) -> str:75 deadline = time.time() + timeout76 last_status = None77 while time.time() < deadline:78 last_status = _get_model_status(model_id)79 if last_status in desired:80 return last_status81 time.sleep(1)82 raise AssertionError(83 f"Timed out waiting for {model_id} to reach {desired}, last status: {last_status}"84 )85 86 87def _load_model_and_wait(88 model_id: str, timeout: int = 60, headers: dict | None = None89) -> None:90 load_res = server.make_request(91 "POST", "/models/load", data={"model": model_id}, headers=headers92 )93 assert load_res.status_code == 20094 assert isinstance(load_res.body, dict)95 assert load_res.body.get("success") is True96 _wait_for_model_status(model_id, {"loaded"}, timeout=timeout)97 98 99def test_router_unload_model():100 global server101 server.start()102 model_id = "ggml-org/tinygemma3-GGUF:Q8_0"103 104 _load_model_and_wait(model_id)105 106 unload_res = server.make_request("POST", "/models/unload", data={"model": model_id})107 assert unload_res.status_code == 200108 assert unload_res.body.get("success") is True109 _wait_for_model_status(model_id, {"unloaded"})110 111 112def test_router_models_max_evicts_lru():113 global server114 server.models_max = 2115 server.start()116 117 candidate_models = [118 "ggml-org/tinygemma3-GGUF:Q8_0",119 "ggml-org/test-model-stories260K:F32",120 "ggml-org/test-model-stories260K-infill:F32",121 ]122 123 # Load only the first 2 models to fill the cache124 first, second, third = candidate_models[:3]125 126 _load_model_and_wait(first, timeout=120)127 _load_model_and_wait(second, timeout=120)128 129 # Verify both models are loaded130 assert _get_model_status(first) == "loaded"131 assert _get_model_status(second) == "loaded"132 133 # Load the third model - this should trigger LRU eviction of the first model134 _load_model_and_wait(third, timeout=120)135 136 # Verify eviction: third is loaded, first was evicted137 assert _get_model_status(third) == "loaded"138 assert _get_model_status(first) == "unloaded"139 140 141def test_router_no_models_autoload():142 global server143 server.no_models_autoload = True144 server.start()145 model_id = "ggml-org/tinygemma3-GGUF:Q8_0"146 147 res = server.make_request(148 "POST",149 "/v1/chat/completions",150 data={151 "model": model_id,152 "messages": [{"role": "user", "content": "hello"}],153 "max_tokens": 4,154 },155 )156 assert res.status_code == 400157 assert "error" in res.body158 159 _load_model_and_wait(model_id)160 161 success_res = server.make_request(162 "POST",163 "/v1/chat/completions",164 data={165 "model": model_id,166 "messages": [{"role": "user", "content": "hello"}],167 "max_tokens": 4,168 },169 )170 assert success_res.status_code == 200171 assert "error" not in success_res.body172 173 174def test_router_api_key_required():175 global server176 server.api_key = "sk-router-secret"177 server.start()178 179 model_id = "ggml-org/tinygemma3-GGUF:Q8_0"180 auth_headers = {"Authorization": f"Bearer {server.api_key}"}181 182 res = server.make_request(183 "POST",184 "/v1/chat/completions",185 data={186 "model": model_id,187 "messages": [{"role": "user", "content": "hello"}],188 "max_tokens": 4,189 },190 )191 assert res.status_code == 401192 assert res.body.get("error", {}).get("type") == "authentication_error"193 194 _load_model_and_wait(model_id, headers=auth_headers)195 196 authed = server.make_request(197 "POST",198 "/v1/chat/completions",199 headers=auth_headers,200 data={201 "model": model_id,202 "messages": [{"role": "user", "content": "hello"}],203 "max_tokens": 4,204 },205 )206 assert authed.status_code == 200207 assert "error" not in authed.body208 