Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1import threading2import pytest3from utils import *4 5server: ServerProcess6 7@pytest.fixture(autouse=True)8def create_server():9 global server10 server = ServerPreset.router()11 12 13def test_router_props():14 global server15 server.models_max = 216 server.no_models_autoload = True17 server.start()18 res = server.make_request("GET", "/props")19 assert res.status_code == 20020 assert res.body["role"] == "router"21 assert res.body["max_instances"] == 222 assert res.body["models_autoload"] is False23 assert res.body["build_info"].startswith("b")24 25 26@pytest.mark.parametrize(27 "model,success",28 [29 ("ggml-org/tinygemma3-GGUF:Q8_0", True),30 ("non-existent/model", False),31 ]32)33def test_router_chat_completion_stream(model: str, success: bool):34 global server35 server.start()36 content = ""37 ex: ServerError | None = None38 try:39 res = server.make_stream_request("POST", "/chat/completions", data={40 "model": model,41 "max_tokens": 16,42 "messages": [43 {"role": "user", "content": "hello"},44 ],45 "stream": True,46 })47 for data in res:48 if data["choices"]:49 choice = data["choices"][0]50 if choice["finish_reason"] in ["stop", "length"]:51 assert "content" not in choice["delta"]52 else:53 assert choice["finish_reason"] is None54 content += choice["delta"]["content"] or ''55 except ServerError as e:56 ex = e57 58 if success:59 assert ex is None60 assert len(content) > 061 else:62 assert ex is not None63 assert content == ""64 65 66def _get_model_ids(is_reload: bool) -> set[str]:67 res = server.make_request("GET", "/models" + ("?reload=1" if is_reload else ""))68 assert res.status_code == 20069 return {item["id"] for item in res.body.get("data", [])}70 71 72def _get_model_status(model_id: str) -> str:73 res = server.make_request("GET", "/models")74 assert res.status_code == 20075 for item in res.body.get("data", []):76 if item.get("id") == model_id or item.get("model") == model_id:77 return item["status"]["value"]78 raise AssertionError(f"Model {model_id} not found in /models response")79 80 81def _wait_for_model_status(model_id: str, desired: set[str], timeout: int = 60) -> str:82 deadline = time.time() + timeout83 last_status = None84 while time.time() < deadline:85 last_status = _get_model_status(model_id)86 if last_status in desired:87 return last_status88 time.sleep(0.01)89 raise AssertionError(90 f"Timed out waiting for {model_id} to reach {desired}, last status: {last_status}"91 )92 93 94def _load_model_and_wait(95 model_id: str, timeout: int = 60, headers: dict | None = None96) -> None:97 load_res = server.make_request(98 "POST", "/models/load", data={"model": model_id}, headers=headers99 )100 assert load_res.status_code == 200101 assert isinstance(load_res.body, dict)102 assert load_res.body.get("success") is True103 _wait_for_model_status(model_id, {"loaded"}, timeout=timeout)104 105 106def test_router_unload_model():107 global server108 server.start()109 model_id = "ggml-org/tinygemma3-GGUF:Q8_0"110 111 _load_model_and_wait(model_id)112 113 unload_res = server.make_request("POST", "/models/unload", data={"model": model_id})114 assert unload_res.status_code == 200115 assert unload_res.body.get("success") is True116 _wait_for_model_status(model_id, {"unloaded"})117 118 119def test_router_models_max_evicts_lru():120 global server121 server.models_max = 2122 server.start()123 124 candidate_models = [125 "ggml-org/tinygemma3-GGUF:Q8_0",126 "ggml-org/test-model-stories260K:F32",127 "ggml-org/test-model-stories260K-infill:F32",128 ]129 130 # Load only the first 2 models to fill the cache131 first, second, third = candidate_models[:3]132 133 _load_model_and_wait(first, timeout=120)134 _load_model_and_wait(second, timeout=120)135 136 # Verify both models are loaded137 assert _get_model_status(first) == "loaded"138 assert _get_model_status(second) == "loaded"139 140 # Load the third model - this should trigger LRU eviction of the first model141 _load_model_and_wait(third, timeout=120)142 143 # Verify eviction: third is loaded, first was evicted144 assert _get_model_status(third) == "loaded"145 assert _get_model_status(first) == "unloaded"146 147 148# server_lru_sched tests (relying on LLAMA_SERVER_DEBUG_FAKE_TIMING)149 150MODEL_A = "ggml-org/tinygemma3-GGUF:Q8_0"151MODEL_B = "ggml-org/test-model-stories260K:F32"152MODEL_C = "ggml-org/test-model-stories260K-infill:F32"153 154 155def _tokenize(model_id: str, timeout: float | None = DEFAULT_REQUEST_TIMEOUT) -> ServerResponse:156 return server.make_request(157 "POST", "/tokenize", data={"model": model_id, "content": "hello world"}, timeout=timeout158 )159 160 161class _Bg:162 """runs one request in a thread, keeps its result, error and finish time"""163 164 def __init__(self, fn):165 self.result = None166 self.error: Exception | None = None167 self.done_at: float = 0.0168 self._thread = threading.Thread(target=self._run, args=(fn,), daemon=True)169 170 def _run(self, fn):171 try:172 self.result = fn()173 except Exception as e:174 self.error = e175 self.done_at = time.time()176 177 def start(self):178 self._thread.start()179 return self180 181 def join(self, timeout: int = 180):182 self._thread.join(timeout)183 assert not self._thread.is_alive(), "background request did not finish in time"184 return self185 186 def assert_ok(self, what: str):187 assert self.error is None, f"{what} raised {self.error!r}"188 assert self.result is not None and self.result.status_code == 200, \189 f"{what} failed: {self.result.status_code if self.result else None} {self.result.body if self.result else None}"190 191 192def test_router_queue_does_not_evict_busy_model():193 """a request that finds no free slot waits, and the model serving a request survives it"""194 global server195 server.models_max = 1196 server.start()197 198 _load_model_and_wait(MODEL_A, timeout=120)199 200 busy = _Bg(lambda: _tokenize(MODEL_A)).start()201 time.sleep(0.5) # let the request reach the child and take the only slot202 203 # no slot free and MODEL_A is busy, so this queues instead of evicting mid-request204 queued = _Bg(lambda: _tokenize(MODEL_B)).start()205 206 busy.join()207 queued.join()208 209 # had MODEL_A been evicted while serving, its own request would have died210 busy.assert_ok("request against the busy model")211 queued.assert_ok("queued request")212 213 _wait_for_model_status(MODEL_B, {"loaded"}, timeout=120)214 assert _get_model_status(MODEL_A) == "unloaded"215 216 217def test_router_queue_coalesces_requests_for_same_model():218 """many requests for one missing model share a slot, so only one model is given up"""219 global server220 server.models_max = 2221 server.start()222 223 _load_model_and_wait(MODEL_A, timeout=120)224 _load_model_and_wait(MODEL_B, timeout=120)225 226 # keep MODEL_A busy so MODEL_B is the only model that can be given up227 busy = _Bg(lambda: _tokenize(MODEL_A)).start()228 time.sleep(0.5)229 230 waiters = [_Bg(lambda: _tokenize(MODEL_C)).start() for _ in range(3)]231 232 busy.join()233 for w in waiters:234 w.join()235 236 busy.assert_ok("request against the busy model")237 for i, w in enumerate(waiters):238 w.assert_ok(f"queued request {i}")239 240 _wait_for_model_status(MODEL_C, {"loaded"}, timeout=120)241 # one entry for 3 requests means one eviction: MODEL_B goes, MODEL_A is left alone.242 # without coalescing the leftover entries still ask for a slot,243 # and MODEL_A is taken too as soon as it goes idle244 assert _get_model_status(MODEL_A) == "loaded"245 assert _get_model_status(MODEL_B) == "unloaded"246 247 248def test_router_queue_client_disconnect_keeps_model():249 """a client that leaves while queued must not cost a running model its slot"""250 global server251 server.models_max = 1252 server.start()253 254 _load_model_and_wait(MODEL_A, timeout=120)255 256 busy = _Bg(lambda: _tokenize(MODEL_A)).start()257 time.sleep(0.5)258 259 # queues behind MODEL_A, then gives up long before MODEL_A goes idle260 with pytest.raises(requests.exceptions.RequestException):261 _tokenize(MODEL_B, timeout=1)262 263 busy.join()264 busy.assert_ok("request against the busy model")265 266 # nobody is waiting anymore, so MODEL_A keeps its slot267 time.sleep(3)268 assert _get_model_status(MODEL_A) == "loaded"269 assert _get_model_status(MODEL_B) == "unloaded"270 271 272def test_router_queue_is_fifo():273 """the queue is served in arrival order"""274 global server275 server.models_max = 1276 server.start()277 278 _load_model_and_wait(MODEL_A, timeout=120)279 280 busy = _Bg(lambda: _tokenize(MODEL_A)).start()281 time.sleep(0.5)282 283 first = _Bg(lambda: _tokenize(MODEL_B)).start()284 time.sleep(1) # keep the arrival order unambiguous285 second = _Bg(lambda: _tokenize(MODEL_C)).start()286 287 busy.join()288 first.join()289 second.join()290 291 busy.assert_ok("request against the busy model")292 first.assert_ok("first queued request")293 second.assert_ok("second queued request")294 295 assert first.done_at < second.done_at, "queue was not served in arrival order"296 297 298def test_router_no_models_autoload():299 global server300 server.no_models_autoload = True301 server.start()302 model_id = "ggml-org/tinygemma3-GGUF:Q8_0"303 304 res = server.make_request(305 "POST",306 "/v1/chat/completions",307 data={308 "model": model_id,309 "messages": [{"role": "user", "content": "hello"}],310 "max_tokens": 4,311 },312 )313 assert res.status_code == 400314 assert "error" in res.body315 316 _load_model_and_wait(model_id)317 318 success_res = server.make_request(319 "POST",320 "/v1/chat/completions",321 data={322 "model": model_id,323 "messages": [{"role": "user", "content": "hello"}],324 "max_tokens": 4,325 },326 )327 assert success_res.status_code == 200328 assert "error" not in success_res.body329 330 331def test_router_api_key_required():332 global server333 server.api_key = "sk-router-secret"334 server.start()335 336 model_id = "ggml-org/tinygemma3-GGUF:Q8_0"337 auth_headers = {"Authorization": f"Bearer {server.api_key}"}338 339 res = server.make_request(340 "POST",341 "/v1/chat/completions",342 data={343 "model": model_id,344 "messages": [{"role": "user", "content": "hello"}],345 "max_tokens": 4,346 },347 )348 assert res.status_code == 401349 assert res.body.get("error", {}).get("type") == "authentication_error"350 351 _load_model_and_wait(model_id, headers=auth_headers)352 353 authed = server.make_request(354 "POST",355 "/v1/chat/completions",356 headers=auth_headers,357 data={358 "model": model_id,359 "messages": [{"role": "user", "content": "hello"}],360 "max_tokens": 4,361 },362 )363 assert authed.status_code == 200364 assert "error" not in authed.body365 366 367def test_router_reload_models():368 """POST /models/reload re-reads the INI preset and updates the model list."""369 global server370 371 preset_path = os.path.join(TMP_DIR, "test_reload.ini")372 373 # Initial preset: two models374 with open(preset_path, "w") as f:375 f.write(376 "[model-reload-a]\n"377 "hf-repo = ggml-org/test-model-stories260K\n"378 "\n"379 "[model-reload-b]\n"380 "hf-repo = ggml-org/test-model-stories260K-infill\n"381 )382 383 server.models_preset = preset_path384 server.start()385 386 ids = _get_model_ids(is_reload=False)387 assert "model-reload-a" in ids388 assert "model-reload-b" in ids389 390 # Updated preset: remove a, keep b unchanged, add c391 with open(preset_path, "w") as f:392 f.write(393 "[model-reload-b]\n"394 "hf-repo = ggml-org/test-model-stories260K-infill\n"395 "\n"396 "[model-reload-c]\n"397 "hf-repo = ggml-org/test-model-stories260K\n"398 )399 400 try:401 ids = _get_model_ids(is_reload=True)402 assert "model-reload-a" not in ids, "removed model should no longer appear"403 assert "model-reload-b" in ids, "unchanged model should still appear"404 assert "model-reload-c" in ids, "newly added model should appear"405 finally:406 os.remove(preset_path)407 408 409def test_router_remote_preset():410 global server411 server.model_hf_repo = "ggml-org/test-preset-ci"412 server.model_hf_file = None413 server.offline = False414 server.start()415 416 # Should see preset models in GET /models417 res = server.make_request("GET", "/models")418 assert res.status_code == 200419 ids = {item["id"] for item in res.body.get("data", [])}420 assert "tinygemma3-preset" in ids421 assert "stories260K-test" in ids422 423 # Should be able to load a preset model424 model_id = "tinygemma3-preset"425 _load_model_and_wait(model_id)426 427 428MODEL_DOWNLOAD_ID = "ggml-org/test-model-router-download:F16"429MODEL_DOWNLOAD_TIMEOUT = 30430 431 432def _listen_sse(433 server: ServerProcess, collected: list, stop: threading.Event, ready: threading.Event | None = None434):435 """Collect /models/sse events into `collected` until `stop` is set.436 437 When `ready` is provided, it is set once the streaming response is open,438 i.e. the server has accepted the connection and registered us as a439 subscriber. Callers that trigger one-shot events (e.g. download_finished)440 must wait on `ready` before acting, otherwise the event can be broadcast441 before this client is subscribed and be lost.442 """443 url = f"http://{server.server_host}:{server.server_port}/models/sse"444 try:445 with requests.get(url, stream=True, timeout=MODEL_DOWNLOAD_TIMEOUT) as resp:446 if ready is not None:447 ready.set()448 for line_bytes in resp.iter_lines():449 if stop.is_set():450 break451 line = line_bytes.decode("utf-8")452 if line.startswith("data: "):453 collected.append(json.loads(line[6:]))454 except Exception:455 pass456 457 458def _wait_for_sse_event(collected: list, event_type: str, model: str, timeout: int) -> bool:459 deadline = time.time() + timeout460 while time.time() < deadline:461 if any(e.get("event") == event_type and e.get("model") == model for e in collected):462 return True463 time.sleep(0.01)464 return False465 466 467def test_router_download_model():468 """Case 1: download a model, verify SSE events and GET /models."""469 global server470 server.start()471 472 # Ensure the model is not present before we start473 server.make_request("DELETE", f"/models?model={MODEL_DOWNLOAD_ID}")474 475 sse_events: list = []476 stop = threading.Event()477 sse_ready = threading.Event()478 sse_thread = threading.Thread(479 target=_listen_sse, args=(server, sse_events, stop, sse_ready), daemon=True480 )481 sse_thread.start()482 483 # wait for the SSE client to be subscribed before triggering the download,484 # otherwise the one-shot download_finished event can be broadcast before485 # this client is registered and be lost486 assert sse_ready.wait(10), "SSE client failed to connect"487 488 # Trigger the download489 res = server.make_request("POST", "/models", data={"model": MODEL_DOWNLOAD_ID})490 assert res.status_code == 200491 assert res.body.get("success") is True492 493 # Wait for download_finished SSE event494 finished = _wait_for_sse_event(495 sse_events, "download_finished", MODEL_DOWNLOAD_ID, MODEL_DOWNLOAD_TIMEOUT496 )497 stop.set()498 499 assert finished, "Never received download_finished SSE event"500 assert any(501 e.get("event") == "download_progress" and e.get("model") == MODEL_DOWNLOAD_ID502 for e in sse_events503 ), "No download_progress events received"504 505 # Model should now appear in GET /models506 ids = _get_model_ids(is_reload=False)507 assert MODEL_DOWNLOAD_ID in ids, f"{MODEL_DOWNLOAD_ID} not found in /models after download"508 509 510def test_router_delete_model():511 """Case 2: delete the downloaded model, verify it disappears from GET /models."""512 global server513 server.start()514 515 # Ensure the model exists (download it if needed)516 if MODEL_DOWNLOAD_ID not in _get_model_ids(is_reload=False):517 sse_events: list = []518 stop = threading.Event()519 sse_ready = threading.Event()520 threading.Thread(521 target=_listen_sse, args=(server, sse_events, stop, sse_ready), daemon=True522 ).start()523 # subscribe before triggering the download so the one-shot524 # download_finished event is not lost (see test_router_download_model)525 assert sse_ready.wait(10), "SSE client failed to connect"526 res = server.make_request("POST", "/models", data={"model": MODEL_DOWNLOAD_ID})527 assert res.status_code == 200528 finished = _wait_for_sse_event(529 sse_events, "download_finished", MODEL_DOWNLOAD_ID, MODEL_DOWNLOAD_TIMEOUT530 )531 stop.set()532 assert finished, "Model did not finish downloading before delete test"533 534 # Delete the model535 del_res = server.make_request("DELETE", f"/models?model={MODEL_DOWNLOAD_ID}")536 assert del_res.status_code == 200537 assert del_res.body.get("success") is True538 539 # Model should no longer appear in GET /models540 ids = _get_model_ids(is_reload=False)541 assert MODEL_DOWNLOAD_ID not in ids, f"{MODEL_DOWNLOAD_ID} still present after deletion"542 