Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1import pytest2from utils import *3 4server = ServerPreset.tinyllama2()5 6 7SHORT_TEXT = """8Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magna aliqua.9Ut enim ad minim veniam, quis nostrud exercitation ullamco laboris nisi ut aliquip ex ea commodo consequat.10Duis aute irure dolor in reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla pariatur.11""".strip()12 13LONG_TEXT = """14Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magna aliqua.15Ut enim ad minim veniam, quis nostrud exercitation ullamco laboris nisi ut aliquip ex ea commodo consequat.16Duis aute irure dolor in reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla pariatur.17Excepteur sint occaecat cupidatat non proident, sunt in culpa qui officia deserunt mollit anim id est laborum.18""".strip()19 20@pytest.fixture(autouse=True)21def create_server():22 global server23 server = ServerPreset.tinyllama2()24 server.n_ctx = 51225 server.n_slots = 226 server.n_predict = 12827 28 29def test_ctx_shift_enabled():30 # the prompt is 226 tokens31 # the slot context is 512/2 = 256 tokens32 # 96 tokens are generated thanks to shifting the context when it gets full33 global server34 server.enable_ctx_shift = True35 server.start()36 res = server.make_request("POST", "/completion", data={37 "n_predict": 96,38 "prompt": SHORT_TEXT,39 })40 assert res.status_code == 20041 assert res.body["timings"]["prompt_n"] == 22642 assert res.body["timings"]["predicted_n"] == 9643 assert res.body["truncated"] is True44 45 46@pytest.mark.parametrize("n_predict,n_token_output,truncated", [47 (64, 64, False),48 (-1, 248, True), # 8 tokens prompt + 248 tokens generated = 256 tokens total49])50def test_ctx_shift_disabled_short_prompt(n_predict: int, n_token_output: int, truncated: bool):51 global server52 server.n_predict = -153 server.start()54 res = server.make_request("POST", "/completion", data={55 "n_predict": n_predict,56 "prompt": "Hi how are you",57 })58 assert res.status_code == 20059 assert res.body["timings"]["predicted_n"] == n_token_output60 assert res.body["truncated"] == truncated61 62 63def test_ctx_shift_disabled_long_prompt():64 global server65 server.start()66 res = server.make_request("POST", "/completion", data={67 "n_predict": 64,68 "prompt": LONG_TEXT,69 })70 assert res.status_code != 20071 assert "error" in res.body72 assert "exceeds the available context size" in res.body["error"]["message"]73 74def test_ctx_shift_disabled_stream():75 global server76 server.start()77 res = server.make_stream_request("POST", "/v1/completions", data={78 "n_predict": 256,79 "prompt": "Once",80 "stream": True,81 })82 content = ""83 for data in res:84 choice = data["choices"][0]85 if choice["finish_reason"] == "length":86 assert len(content) > 087 else:88 assert choice["finish_reason"] is None89 content += choice["text"]90 