Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
test_ctx_shift.py68 linesDownload Raw Back to unit
1import pytest2from utils import *3 4server = ServerPreset.tinyllama2()5 6 7LONG_TEXT = """8Lorem ipsum dolor sit amet, consectetur adipiscing elit, sed do eiusmod tempor incididunt ut labore et dolore magna aliqua.9Ut enim ad minim veniam, quis nostrud exercitation ullamco laboris nisi ut aliquip ex ea commodo consequat.10Duis aute irure dolor in reprehenderit in voluptate velit esse cillum dolore eu fugiat nulla pariatur.11Excepteur sint occaecat cupidatat non proident, sunt in culpa qui officia deserunt mollit anim id est laborum.12""".strip()13 14@pytest.fixture(scope="module", autouse=True)15def create_server():16    global server17    server = ServerPreset.tinyllama2()18    server.n_ctx = 25619    server.n_slots = 220 21 22def test_ctx_shift_enabled():23    # the prompt is 301 tokens24    # the slot context is 256/2 = 128 tokens25    # the prompt is truncated to keep the last 109 tokens26    # 64 tokens are generated thanks to shifting the context when it gets full27    global server28    server.start()29    res = server.make_request("POST", "/completion", data={30        "n_predict": 64,31        "prompt": LONG_TEXT,32    })33    assert res.status_code == 20034    assert res.body["timings"]["prompt_n"] == 10935    assert res.body["timings"]["predicted_n"] == 6436    assert res.body["truncated"] is True37 38 39@pytest.mark.parametrize("n_predict,n_token_output,truncated", [40    (64, 64, False),41    (-1, 120, True),42])43def test_ctx_shift_disabled_short_prompt(n_predict: int, n_token_output: int, truncated: bool):44    global server45    server.disable_ctx_shift = True46    server.n_predict = -147    server.start()48    res = server.make_request("POST", "/completion", data={49        "n_predict": n_predict,50        "prompt": "Hi how are you",51    })52    assert res.status_code == 20053    assert res.body["timings"]["predicted_n"] == n_token_output54    assert res.body["truncated"] == truncated55 56 57def test_ctx_shift_disabled_long_prompt():58    global server59    server.disable_ctx_shift = True60    server.start()61    res = server.make_request("POST", "/completion", data={62        "n_predict": 64,63        "prompt": LONG_TEXT,64    })65    assert res.status_code != 20066    assert "error" in res.body67    assert "exceeds the available context size" in res.body["error"]["message"]68