Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
test_kv_keep_only_active.py116 linesDownload Raw Back to unit
1import os2import tempfile3import pytest4from utils import *5 6server = ServerPreset.tinyllama2()7 8class LogReader:9    def __init__(self, path):10        self.path = path11        self.pos = 012    def drain(self):13        with open(self.path) as f:14            f.seek(self.pos)15            content = f.read()16            self.pos = f.tell()17        return content18 19@pytest.fixture(autouse=True)20def create_server():21    global server22    server = ServerPreset.tinyllama2()23    server.n_slots = 224    server.n_predict = 425    server.temperature = 0.026    server.server_slots = True27    server.cache_ram = 10028    server.kv_unified = True29    server.debug = True30    fd, server.log_path = tempfile.mkstemp(suffix='.log')31    os.close(fd)32    yield33 34 35LONG_PROMPT = (36    "Once upon a time in a land far away, there lived a brave knight "37    "who traveled across mountains and rivers to find the legendary "38    "golden sword hidden deep within the enchanted forest of whispers. "39    "He met many creatures along the way including dragons and fairies "40    "and wizards who helped him on his noble quest to save the kingdom."41)42 43 44# idle slot cleared on launch should restore from cache-ram45def test_clear_and_restore():46    global server47    server.start()48    log = LogReader(server.log_path)49 50    # verify feature is enabled51    assert "__TEST_TAG_CACHE_IDLE_SLOTS_ENABLED__" in log.drain()52 53    res = server.make_request("POST", "/completion", data={54        "prompt": LONG_PROMPT,55        "id_slot": 0,56        "cache_prompt": True,57    })58    assert res.status_code == 20059    original_prompt_n = res.body["timings"]["prompt_n"]60 61    # Slot 0 is the only slot with KV — should NOT be cleared62    assert "__TEST_TAG_CACHE_IDLE_SLOT__" not in log.drain()63 64    # Launching slot 1 clears idle slot 065    res = server.make_request("POST", "/completion", data={66        "prompt": "The quick brown fox",67        "id_slot": 1,68        "cache_prompt": True,69    })70    assert res.status_code == 20071    assert "__TEST_TAG_CACHE_IDLE_SLOT__" in log.drain()72 73    # Re-send same prompt — should restore from cache-ram74    res = server.make_request("POST", "/completion", data={75        "prompt": LONG_PROMPT,76        "cache_prompt": True,77    })78    assert res.status_code == 20079    assert "updating prompt cache" in log.drain()80    assert res.body["timings"]["cache_n"] > 081    assert res.body["timings"]["prompt_n"] < original_prompt_n82 83    # Follow-up — slot 0 kept its KV, no clearing needed84    res = server.make_request("POST", "/completion", data={85        "prompt": LONG_PROMPT + " The knight finally reached the castle gates.",86        "cache_prompt": True,87    })88    assert res.status_code == 20089    assert "__TEST_TAG_CACHE_IDLE_SLOT__" not in log.drain()90 91 92def test_disabled_with_flag():93    global server94    server.no_cache_idle_slots = True95    server.start()96    log = LogReader(server.log_path)97 98    # Feature should not be enabled99    assert "__TEST_TAG_CACHE_IDLE_SLOTS_ENABLED__" not in log.drain()100 101    res = server.make_request("POST", "/completion", data={102        "prompt": LONG_PROMPT,103        "id_slot": 0,104        "cache_prompt": True,105    })106    assert res.status_code == 200107 108    # Request on different slot — should NOT trigger clearing109    res = server.make_request("POST", "/completion", data={110        "prompt": "The quick brown fox",111        "id_slot": 1,112        "cache_prompt": True,113    })114    assert res.status_code == 200115    assert "__TEST_TAG_CACHE_IDLE_SLOT__" not in log.drain()116 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai