Felipe97/llama-cpp-compiled
01.2k
1import pytest2from utils import *3 4server: ServerProcess5 6 7@pytest.fixture(autouse=True)8def create_server():9 global server10 server = ServerPreset.tinyllama2()11 server.gcp_compat = True12 13 14def test_gcp_predict_camel_case():15 global server16 server.start()17 res = server.make_request("POST", "/predict", data={18 "instances": [19 {20 "@requestFormat": "chatCompletions",21 "max_tokens": 8,22 "messages": [23 {"role": "user", "content": "What is the meaning of life?"},24 ],25 }26 ],27 })28 assert res.status_code == 20029 assert "predictions" in res.body30 assert len(res.body["predictions"]) == 131 prediction = res.body["predictions"][0]32 assert "choices" in prediction33 assert len(prediction["choices"]) == 134 assert prediction["choices"][0]["message"]["role"] == "assistant"35 assert len(prediction["choices"][0]["message"]["content"]) > 036 37 38def test_gcp_predict_multiple_instances():39 global server40 server.n_slots = 241 server.start()42 res = server.make_request("POST", "/predict", data={43 "instances": [44 {45 "@requestFormat": "chatCompletions",46 "max_tokens": 8,47 "messages": [{"role": "user", "content": "Say hello"}],48 },49 {50 "@requestFormat": "chatCompletions",51 "max_tokens": 8,52 "messages": [{"role": "user", "content": "Say world"}],53 },54 ],55 })56 assert res.status_code == 20057 assert len(res.body["predictions"]) == 258 for prediction in res.body["predictions"]:59 assert "choices" in prediction60 assert len(prediction["choices"][0]["message"]["content"]) > 061 