Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
test_router.py542 linesDownload Raw Back to unit
1import threading2import pytest3from utils import *4 5server: ServerProcess6 7@pytest.fixture(autouse=True)8def create_server():9    global server10    server = ServerPreset.router()11 12 13def test_router_props():14    global server15    server.models_max = 216    server.no_models_autoload = True17    server.start()18    res = server.make_request("GET", "/props")19    assert res.status_code == 20020    assert res.body["role"] == "router"21    assert res.body["max_instances"] == 222    assert res.body["models_autoload"] is False23    assert res.body["build_info"].startswith("b")24 25 26@pytest.mark.parametrize(27    "model,success",28    [29        ("ggml-org/tinygemma3-GGUF:Q8_0", True),30        ("non-existent/model", False),31    ]32)33def test_router_chat_completion_stream(model: str, success: bool):34    global server35    server.start()36    content = ""37    ex: ServerError | None = None38    try:39        res = server.make_stream_request("POST", "/chat/completions", data={40            "model": model,41            "max_tokens": 16,42            "messages": [43                {"role": "user", "content": "hello"},44            ],45            "stream": True,46        })47        for data in res:48            if data["choices"]:49                choice = data["choices"][0]50                if choice["finish_reason"] in ["stop", "length"]:51                    assert "content" not in choice["delta"]52                else:53                    assert choice["finish_reason"] is None54                    content += choice["delta"]["content"] or ''55    except ServerError as e:56        ex = e57 58    if success:59        assert ex is None60        assert len(content) > 061    else:62        assert ex is not None63        assert content == ""64 65 66def _get_model_ids(is_reload: bool) -> set[str]:67    res = server.make_request("GET", "/models" + ("?reload=1" if is_reload else ""))68    assert res.status_code == 20069    return {item["id"] for item in res.body.get("data", [])}70 71 72def _get_model_status(model_id: str) -> str:73    res = server.make_request("GET", "/models")74    assert res.status_code == 20075    for item in res.body.get("data", []):76        if item.get("id") == model_id or item.get("model") == model_id:77            return item["status"]["value"]78    raise AssertionError(f"Model {model_id} not found in /models response")79 80 81def _wait_for_model_status(model_id: str, desired: set[str], timeout: int = 60) -> str:82    deadline = time.time() + timeout83    last_status = None84    while time.time() < deadline:85        last_status = _get_model_status(model_id)86        if last_status in desired:87            return last_status88        time.sleep(0.01)89    raise AssertionError(90        f"Timed out waiting for {model_id} to reach {desired}, last status: {last_status}"91    )92 93 94def _load_model_and_wait(95    model_id: str, timeout: int = 60, headers: dict | None = None96) -> None:97    load_res = server.make_request(98        "POST", "/models/load", data={"model": model_id}, headers=headers99    )100    assert load_res.status_code == 200101    assert isinstance(load_res.body, dict)102    assert load_res.body.get("success") is True103    _wait_for_model_status(model_id, {"loaded"}, timeout=timeout)104 105 106def test_router_unload_model():107    global server108    server.start()109    model_id = "ggml-org/tinygemma3-GGUF:Q8_0"110 111    _load_model_and_wait(model_id)112 113    unload_res = server.make_request("POST", "/models/unload", data={"model": model_id})114    assert unload_res.status_code == 200115    assert unload_res.body.get("success") is True116    _wait_for_model_status(model_id, {"unloaded"})117 118 119def test_router_models_max_evicts_lru():120    global server121    server.models_max = 2122    server.start()123 124    candidate_models = [125        "ggml-org/tinygemma3-GGUF:Q8_0",126        "ggml-org/test-model-stories260K:F32",127        "ggml-org/test-model-stories260K-infill:F32",128    ]129 130    # Load only the first 2 models to fill the cache131    first, second, third = candidate_models[:3]132 133    _load_model_and_wait(first, timeout=120)134    _load_model_and_wait(second, timeout=120)135 136    # Verify both models are loaded137    assert _get_model_status(first) == "loaded"138    assert _get_model_status(second) == "loaded"139 140    # Load the third model - this should trigger LRU eviction of the first model141    _load_model_and_wait(third, timeout=120)142 143    # Verify eviction: third is loaded, first was evicted144    assert _get_model_status(third) == "loaded"145    assert _get_model_status(first) == "unloaded"146 147 148# server_lru_sched tests (relying on LLAMA_SERVER_DEBUG_FAKE_TIMING)149 150MODEL_A = "ggml-org/tinygemma3-GGUF:Q8_0"151MODEL_B = "ggml-org/test-model-stories260K:F32"152MODEL_C = "ggml-org/test-model-stories260K-infill:F32"153 154 155def _tokenize(model_id: str, timeout: float | None = DEFAULT_REQUEST_TIMEOUT) -> ServerResponse:156    return server.make_request(157        "POST", "/tokenize", data={"model": model_id, "content": "hello world"}, timeout=timeout158    )159 160 161class _Bg:162    """runs one request in a thread, keeps its result, error and finish time"""163 164    def __init__(self, fn):165        self.result = None166        self.error: Exception | None = None167        self.done_at: float = 0.0168        self._thread = threading.Thread(target=self._run, args=(fn,), daemon=True)169 170    def _run(self, fn):171        try:172            self.result = fn()173        except Exception as e:174            self.error = e175        self.done_at = time.time()176 177    def start(self):178        self._thread.start()179        return self180 181    def join(self, timeout: int = 180):182        self._thread.join(timeout)183        assert not self._thread.is_alive(), "background request did not finish in time"184        return self185 186    def assert_ok(self, what: str):187        assert self.error is None, f"{what} raised {self.error!r}"188        assert self.result is not None and self.result.status_code == 200, \189            f"{what} failed: {self.result.status_code if self.result else None} {self.result.body if self.result else None}"190 191 192def test_router_queue_does_not_evict_busy_model():193    """a request that finds no free slot waits, and the model serving a request survives it"""194    global server195    server.models_max = 1196    server.start()197 198    _load_model_and_wait(MODEL_A, timeout=120)199 200    busy = _Bg(lambda: _tokenize(MODEL_A)).start()201    time.sleep(0.5)  # let the request reach the child and take the only slot202 203    # no slot free and MODEL_A is busy, so this queues instead of evicting mid-request204    queued = _Bg(lambda: _tokenize(MODEL_B)).start()205 206    busy.join()207    queued.join()208 209    # had MODEL_A been evicted while serving, its own request would have died210    busy.assert_ok("request against the busy model")211    queued.assert_ok("queued request")212 213    _wait_for_model_status(MODEL_B, {"loaded"}, timeout=120)214    assert _get_model_status(MODEL_A) == "unloaded"215 216 217def test_router_queue_coalesces_requests_for_same_model():218    """many requests for one missing model share a slot, so only one model is given up"""219    global server220    server.models_max = 2221    server.start()222 223    _load_model_and_wait(MODEL_A, timeout=120)224    _load_model_and_wait(MODEL_B, timeout=120)225 226    # keep MODEL_A busy so MODEL_B is the only model that can be given up227    busy = _Bg(lambda: _tokenize(MODEL_A)).start()228    time.sleep(0.5)229 230    waiters = [_Bg(lambda: _tokenize(MODEL_C)).start() for _ in range(3)]231 232    busy.join()233    for w in waiters:234        w.join()235 236    busy.assert_ok("request against the busy model")237    for i, w in enumerate(waiters):238        w.assert_ok(f"queued request {i}")239 240    _wait_for_model_status(MODEL_C, {"loaded"}, timeout=120)241    # one entry for 3 requests means one eviction: MODEL_B goes, MODEL_A is left alone.242    # without coalescing the leftover entries still ask for a slot,243    # and MODEL_A is taken too as soon as it goes idle244    assert _get_model_status(MODEL_A) == "loaded"245    assert _get_model_status(MODEL_B) == "unloaded"246 247 248def test_router_queue_client_disconnect_keeps_model():249    """a client that leaves while queued must not cost a running model its slot"""250    global server251    server.models_max = 1252    server.start()253 254    _load_model_and_wait(MODEL_A, timeout=120)255 256    busy = _Bg(lambda: _tokenize(MODEL_A)).start()257    time.sleep(0.5)258 259    # queues behind MODEL_A, then gives up long before MODEL_A goes idle260    with pytest.raises(requests.exceptions.RequestException):261        _tokenize(MODEL_B, timeout=1)262 263    busy.join()264    busy.assert_ok("request against the busy model")265 266    # nobody is waiting anymore, so MODEL_A keeps its slot267    time.sleep(3)268    assert _get_model_status(MODEL_A) == "loaded"269    assert _get_model_status(MODEL_B) == "unloaded"270 271 272def test_router_queue_is_fifo():273    """the queue is served in arrival order"""274    global server275    server.models_max = 1276    server.start()277 278    _load_model_and_wait(MODEL_A, timeout=120)279 280    busy = _Bg(lambda: _tokenize(MODEL_A)).start()281    time.sleep(0.5)282 283    first = _Bg(lambda: _tokenize(MODEL_B)).start()284    time.sleep(1)  # keep the arrival order unambiguous285    second = _Bg(lambda: _tokenize(MODEL_C)).start()286 287    busy.join()288    first.join()289    second.join()290 291    busy.assert_ok("request against the busy model")292    first.assert_ok("first queued request")293    second.assert_ok("second queued request")294 295    assert first.done_at < second.done_at, "queue was not served in arrival order"296 297 298def test_router_no_models_autoload():299    global server300    server.no_models_autoload = True301    server.start()302    model_id = "ggml-org/tinygemma3-GGUF:Q8_0"303 304    res = server.make_request(305        "POST",306        "/v1/chat/completions",307        data={308            "model": model_id,309            "messages": [{"role": "user", "content": "hello"}],310            "max_tokens": 4,311        },312    )313    assert res.status_code == 400314    assert "error" in res.body315 316    _load_model_and_wait(model_id)317 318    success_res = server.make_request(319        "POST",320        "/v1/chat/completions",321        data={322            "model": model_id,323            "messages": [{"role": "user", "content": "hello"}],324            "max_tokens": 4,325        },326    )327    assert success_res.status_code == 200328    assert "error" not in success_res.body329 330 331def test_router_api_key_required():332    global server333    server.api_key = "sk-router-secret"334    server.start()335 336    model_id = "ggml-org/tinygemma3-GGUF:Q8_0"337    auth_headers = {"Authorization": f"Bearer {server.api_key}"}338 339    res = server.make_request(340        "POST",341        "/v1/chat/completions",342        data={343            "model": model_id,344            "messages": [{"role": "user", "content": "hello"}],345            "max_tokens": 4,346        },347    )348    assert res.status_code == 401349    assert res.body.get("error", {}).get("type") == "authentication_error"350 351    _load_model_and_wait(model_id, headers=auth_headers)352 353    authed = server.make_request(354        "POST",355        "/v1/chat/completions",356        headers=auth_headers,357        data={358            "model": model_id,359            "messages": [{"role": "user", "content": "hello"}],360            "max_tokens": 4,361        },362    )363    assert authed.status_code == 200364    assert "error" not in authed.body365 366 367def test_router_reload_models():368    """POST /models/reload re-reads the INI preset and updates the model list."""369    global server370 371    preset_path = os.path.join(TMP_DIR, "test_reload.ini")372 373    # Initial preset: two models374    with open(preset_path, "w") as f:375        f.write(376            "[model-reload-a]\n"377            "hf-repo = ggml-org/test-model-stories260K\n"378            "\n"379            "[model-reload-b]\n"380            "hf-repo = ggml-org/test-model-stories260K-infill\n"381        )382 383    server.models_preset = preset_path384    server.start()385 386    ids = _get_model_ids(is_reload=False)387    assert "model-reload-a" in ids388    assert "model-reload-b" in ids389 390    # Updated preset: remove a, keep b unchanged, add c391    with open(preset_path, "w") as f:392        f.write(393            "[model-reload-b]\n"394            "hf-repo = ggml-org/test-model-stories260K-infill\n"395            "\n"396            "[model-reload-c]\n"397            "hf-repo = ggml-org/test-model-stories260K\n"398        )399 400    try:401        ids = _get_model_ids(is_reload=True)402        assert "model-reload-a" not in ids, "removed model should no longer appear"403        assert "model-reload-b" in ids, "unchanged model should still appear"404        assert "model-reload-c" in ids, "newly added model should appear"405    finally:406        os.remove(preset_path)407 408 409def test_router_remote_preset():410    global server411    server.model_hf_repo = "ggml-org/test-preset-ci"412    server.model_hf_file = None413    server.offline = False414    server.start()415 416    # Should see preset models in GET /models417    res = server.make_request("GET", "/models")418    assert res.status_code == 200419    ids = {item["id"] for item in res.body.get("data", [])}420    assert "tinygemma3-preset" in ids421    assert "stories260K-test" in ids422 423    # Should be able to load a preset model424    model_id = "tinygemma3-preset"425    _load_model_and_wait(model_id)426 427 428MODEL_DOWNLOAD_ID = "ggml-org/test-model-router-download:F16"429MODEL_DOWNLOAD_TIMEOUT = 30430 431 432def _listen_sse(433    server: ServerProcess, collected: list, stop: threading.Event, ready: threading.Event | None = None434):435    """Collect /models/sse events into `collected` until `stop` is set.436 437    When `ready` is provided, it is set once the streaming response is open,438    i.e. the server has accepted the connection and registered us as a439    subscriber. Callers that trigger one-shot events (e.g. download_finished)440    must wait on `ready` before acting, otherwise the event can be broadcast441    before this client is subscribed and be lost.442    """443    url = f"http://{server.server_host}:{server.server_port}/models/sse"444    try:445        with requests.get(url, stream=True, timeout=MODEL_DOWNLOAD_TIMEOUT) as resp:446            if ready is not None:447                ready.set()448            for line_bytes in resp.iter_lines():449                if stop.is_set():450                    break451                line = line_bytes.decode("utf-8")452                if line.startswith("data: "):453                    collected.append(json.loads(line[6:]))454    except Exception:455        pass456 457 458def _wait_for_sse_event(collected: list, event_type: str, model: str, timeout: int) -> bool:459    deadline = time.time() + timeout460    while time.time() < deadline:461        if any(e.get("event") == event_type and e.get("model") == model for e in collected):462            return True463        time.sleep(0.01)464    return False465 466 467def test_router_download_model():468    """Case 1: download a model, verify SSE events and GET /models."""469    global server470    server.start()471 472    # Ensure the model is not present before we start473    server.make_request("DELETE", f"/models?model={MODEL_DOWNLOAD_ID}")474 475    sse_events: list = []476    stop = threading.Event()477    sse_ready = threading.Event()478    sse_thread = threading.Thread(479        target=_listen_sse, args=(server, sse_events, stop, sse_ready), daemon=True480    )481    sse_thread.start()482 483    # wait for the SSE client to be subscribed before triggering the download,484    # otherwise the one-shot download_finished event can be broadcast before485    # this client is registered and be lost486    assert sse_ready.wait(10), "SSE client failed to connect"487 488    # Trigger the download489    res = server.make_request("POST", "/models", data={"model": MODEL_DOWNLOAD_ID})490    assert res.status_code == 200491    assert res.body.get("success") is True492 493    # Wait for download_finished SSE event494    finished = _wait_for_sse_event(495        sse_events, "download_finished", MODEL_DOWNLOAD_ID, MODEL_DOWNLOAD_TIMEOUT496    )497    stop.set()498 499    assert finished, "Never received download_finished SSE event"500    assert any(501        e.get("event") == "download_progress" and e.get("model") == MODEL_DOWNLOAD_ID502        for e in sse_events503    ), "No download_progress events received"504 505    # Model should now appear in GET /models506    ids = _get_model_ids(is_reload=False)507    assert MODEL_DOWNLOAD_ID in ids, f"{MODEL_DOWNLOAD_ID} not found in /models after download"508 509 510def test_router_delete_model():511    """Case 2: delete the downloaded model, verify it disappears from GET /models."""512    global server513    server.start()514 515    # Ensure the model exists (download it if needed)516    if MODEL_DOWNLOAD_ID not in _get_model_ids(is_reload=False):517        sse_events: list = []518        stop = threading.Event()519        sse_ready = threading.Event()520        threading.Thread(521            target=_listen_sse, args=(server, sse_events, stop, sse_ready), daemon=True522        ).start()523        # subscribe before triggering the download so the one-shot524        # download_finished event is not lost (see test_router_download_model)525        assert sse_ready.wait(10), "SSE client failed to connect"526        res = server.make_request("POST", "/models", data={"model": MODEL_DOWNLOAD_ID})527        assert res.status_code == 200528        finished = _wait_for_sse_event(529            sse_events, "download_finished", MODEL_DOWNLOAD_ID, MODEL_DOWNLOAD_TIMEOUT530        )531        stop.set()532        assert finished, "Model did not finish downloading before delete test"533 534    # Delete the model535    del_res = server.make_request("DELETE", f"/models?model={MODEL_DOWNLOAD_ID}")536    assert del_res.status_code == 200537    assert del_res.body.get("success") is True538 539    # Model should no longer appear in GET /models540    ids = _get_model_ids(is_reload=False)541    assert MODEL_DOWNLOAD_ID not in ids, f"{MODEL_DOWNLOAD_ID} still present after deletion"542 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai