Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#pragma once2 3#include "common.h"4#include "llama.h"5 6#include <string>7#include <unordered_set>8#include <list>9#include <map>10 11// TODO: prevent including the whole server-common.h as we only use server_tokens12#include "server-common.h"13 14using json = nlohmann::ordered_json;15 16enum server_task_type {17 SERVER_TASK_TYPE_COMPLETION,18 SERVER_TASK_TYPE_EMBEDDING,19 SERVER_TASK_TYPE_RERANK,20 SERVER_TASK_TYPE_INFILL,21 SERVER_TASK_TYPE_CANCEL,22 SERVER_TASK_TYPE_CONTROL,23 SERVER_TASK_TYPE_NEXT_RESPONSE,24 SERVER_TASK_TYPE_METRICS,25 SERVER_TASK_TYPE_SLOT_SAVE,26 SERVER_TASK_TYPE_SLOT_RESTORE,27 SERVER_TASK_TYPE_SLOT_ERASE,28 SERVER_TASK_TYPE_GET_LORA,29 SERVER_TASK_TYPE_SET_LORA,30};31 32// TODO: change this to more generic "response_format" to replace the "format_response_*" in server-common33enum task_response_type {34 TASK_RESPONSE_TYPE_NONE, // llama.cpp native format35 TASK_RESPONSE_TYPE_OAI_CHAT,36 TASK_RESPONSE_TYPE_OAI_CMPL,37 TASK_RESPONSE_TYPE_OAI_RESP,38 TASK_RESPONSE_TYPE_OAI_ASR, // transcriptions API39 TASK_RESPONSE_TYPE_OAI_EMBD,40 TASK_RESPONSE_TYPE_ANTHROPIC,41};42 43enum stop_type {44 STOP_TYPE_NONE,45 STOP_TYPE_EOS,46 STOP_TYPE_WORD,47 STOP_TYPE_LIMIT,48};49 50struct task_params {51 bool stream = false;52 bool include_usage = false;53 bool cache_prompt = true; // remember the prompt to avoid reprocessing all prompt54 bool return_tokens = false;55 bool return_progress = false;56 57 int32_t sse_ping_interval = 30; // seconds between SSE comment pings while the stream stays silent, -1 disables58 59 int32_t n_keep = 0; // number of tokens to keep from initial prompt60 int32_t n_discard = 0; // number of tokens after n_keep that may be discarded when shifting context, 0 defaults to half61 int32_t n_predict = -1; // new tokens to predict62 int32_t n_indent = 0; // minimum line indentation for the generated text in number of whitespace characters63 int32_t n_cmpl = 1; // number of completions to generate from this prompt64 65 int32_t n_cache_reuse = 0; // min chunk size to attempt reusing from the cache via KV shifting (0 = disabled)66 67 int64_t t_max_prompt_ms = -1; // TODO: implement68 int64_t t_max_predict_ms = -1; // if positive, limit the generation phase to this time limit69 70 std::map<int, float> lora; // mapping adapter ID -> scale71 72 std::vector<std::string> antiprompt;73 std::vector<std::string> response_fields;74 75 bool timings_per_token = false;76 bool post_sampling_probs = false;77 78 struct common_params_sampling sampling;79 struct common_params_speculative speculative;80 81 // response formatting82 bool verbose = false;83 task_response_type res_type = TASK_RESPONSE_TYPE_NONE;84 std::string oaicompat_model;85 std::string oaicompat_cmpl_id;86 87 // realtime control (SERVER_TASK_TYPE_CONTROL)88 std::string control_action;89 std::string control_cmpl_id;90 91 // per-request parameters for chat parsing92 common_chat_parser_params chat_parser_params;93 94 // message spans for checkpointing95 common_chat_msg_spans message_spans;96 97 // Embeddings98 int32_t embd_normalize = 2; // (-1=none, 0=max absolute int16, 1=taxicab, 2=Euclidean/L2, >2=p-norm)99 100 json format_logit_bias(const std::vector<llama_logit_bias> & logit_bias) const;101 json to_json(bool only_metrics = false) const;102};103 104// struct for tracking the state of a task (e.g., for streaming)105struct task_result_state {106 // tracking diffs for partial tool calls107 std::vector<common_chat_msg_diff> diffs;108 common_chat_parser_params chat_parser_params;109 common_chat_msg chat_msg;110 std::string generated_text; // append new chunks of generated text here111 std::vector<std::string> generated_tool_call_ids;112 std::unordered_set<size_t> sent_tool_call_names;113 114 // for OpenAI Responses and Anthropic streaming API:115 // track output item / content block state across chunks116 bool thinking_block_started = false;117 bool text_block_started = false;118 119 // for OpenAI Responses streaming API120 bool oai_resp_created = false;121 const std::string oai_resp_id;122 const std::string oai_resp_reasoning_id;123 const std::string oai_resp_message_id;124 std::string oai_resp_fc_id; // function call ID for current args delta125 126 task_result_state(const common_chat_parser_params & chat_parser_params);127 128 // parse partial tool calls and update the internal state129 common_chat_msg update_chat_msg(130 const std::string & text_added,131 bool is_partial,132 std::vector<common_chat_msg_diff> & diffs,133 bool filter_tool_calls = false);134};135 136struct server_task {137 int id = -1; // to be filled by server_queue138 139 // TODO @ngxson : remove this field and implement a mapping task_id -> idx in the response_reader140 size_t index = 0; // used when there are multiple prompts (batch request)141 142 // used by SERVER_TASK_TYPE_CANCEL143 int id_target = -1;144 int id_slot = -1;145 146 // used by parallel sampling (multiple completions from same prompt)147 int id_parent = -1;148 // temporary store of child tasks for scheduling149 // note: accessing to elements is invalid after the task is moved to server_slot150 std::vector<server_task> child_tasks;151 152 // used by SERVER_TASK_TYPE_INFERENCE153 task_params params;154 server_tokens tokens;155 156 // only used by CLI, this allow tokenizing CLI inputs on server side157 // we need this because mtmd_context and vocab are not accessible outside of server_context158 bool cli = false;159 std::string cli_prompt;160 std::vector<raw_buffer> cli_files;161 162 server_task_type type;163 164 // used by SERVER_TASK_TYPE_SLOT_SAVE, SERVER_TASK_TYPE_SLOT_RESTORE, SERVER_TASK_TYPE_SLOT_ERASE165 struct slot_action {166 int id_slot;167 std::string filename;168 std::string filepath;169 };170 slot_action slot_action;171 172 // used by SERVER_TASK_TYPE_METRICS173 bool metrics_reset_bucket = false;174 175 // used by SERVER_TASK_TYPE_SET_LORA176 std::map<int, float> set_lora; // mapping adapter ID -> scale177 178 server_task() = default;179 180 server_task(server_task_type type) : type(type) {}181 182 int32_t n_tokens() const {183 return tokens.size();184 }185 186 bool need_embd() const {187 switch (type) {188 case SERVER_TASK_TYPE_EMBEDDING:189 case SERVER_TASK_TYPE_RERANK:190 return true;191 default:192 return false;193 }194 }195 196 bool need_logits() const {197 switch (type) {198 case SERVER_TASK_TYPE_COMPLETION:199 case SERVER_TASK_TYPE_INFILL:200 return true;201 default:202 return false;203 }204 }205 206 bool need_sampling() const {207 switch (type) {208 case SERVER_TASK_TYPE_COMPLETION:209 case SERVER_TASK_TYPE_INFILL:210 return true;211 default:212 return false;213 }214 }215 216 // utility function217 static std::unordered_set<int> get_list_id(const std::vector<server_task> & tasks) {218 std::unordered_set<int> ids(tasks.size());219 for (size_t i = 0; i < tasks.size(); i++) {220 ids.insert(tasks[i].id);221 for (auto & child : tasks[i].child_tasks) {222 ids.insert(child.id);223 }224 }225 return ids;226 }227 228 void add_child(int id_parent, int id_child) {229 server_task copy;230 231 copy.id = id_child;232 copy.id_parent = id_parent;233 copy.params = params;234 copy.type = type;235 copy.tokens = tokens.clone();236 copy.id_slot = -1; // child tasks cannot specify slot237 238 // use different sampling seed for each child239 // note: https://github.com/ggml-org/llama.cpp/pull/18700#discussion_r2675115723240 if (copy.params.sampling.seed != LLAMA_DEFAULT_SEED) {241 copy.params.sampling.seed += (uint32_t)child_tasks.size() + 1;242 }243 244 child_tasks.push_back(std::move(copy));245 }246 247 // the task will be moved into queue, then onto slots248 // however, the state must be kept by caller (e.g., HTTP thread)249 task_result_state create_state() const {250 return task_result_state(params.chat_parser_params);251 }252 253 bool is_parent() const {254 return child_tasks.size() > 0;255 }256 257 bool is_child() const {258 return id_parent != -1;259 }260};261 262struct result_timings {263 int32_t cache_n = -1;264 265 int32_t prompt_n = -1;266 double prompt_ms = 0.0;267 double prompt_per_token_ms = 0.0;268 double prompt_per_second = 0.0;269 270 int32_t predicted_n = -1;271 double predicted_ms = 0.0;272 double predicted_per_token_ms = 0.0;273 double predicted_per_second = 0.0;274 275 // Optional speculative metrics - only included when > 0276 int32_t draft_n = 0;277 int32_t draft_n_accepted = 0;278 279 json to_json() const;280};281 282struct result_prompt_progress {283 int32_t total = 0;284 int32_t cache = 0;285 int32_t processed = 0;286 int64_t time_ms = 0;287 288 json to_json() const;289};290 291struct server_task_result {292 int id = -1;293 int id_slot = -1;294 295 // TODO @ngxson : remove this field and implement a mapping task_id -> idx in the response_reader296 size_t index = 0; // to be used for batched tasks297 298 virtual bool is_error() {299 // only used by server_task_result_error300 return false;301 }302 virtual bool is_stop() {303 // only used by server_task_result_cmpl_*304 return true;305 }306 virtual void update(task_result_state &) {307 // only used by server_task_result_cmpl_*308 }309 virtual json to_json() = 0;310 virtual ~server_task_result() = default;311 virtual server_task_result * clone() const {312 GGML_ABORT("not implemented for this task type");313 }314};315 316// using shared_ptr for polymorphism of server_task_result317using server_task_result_ptr = std::unique_ptr<server_task_result>;318 319struct completion_token_output {320 llama_token tok;321 float prob;322 std::string text_to_send;323 struct prob_info {324 llama_token tok;325 std::string txt;326 float prob;327 };328 std::vector<prob_info> probs;329 330 json to_json(bool post_sampling_probs) const;331 332 static json probs_vector_to_json(const std::vector<completion_token_output> & probs, bool post_sampling_probs);333 334 static float logarithm(float x);335 336 static std::vector<unsigned char> str_to_bytes(const std::string & str);337 338};339 340struct server_task_result_cmpl_final : server_task_result {341 std::string content;342 llama_tokens tokens;343 344 bool stream;345 bool include_usage;346 result_timings timings;347 std::string prompt;348 349 bool truncated;350 int32_t n_decoded;351 int32_t n_prompt_tokens;352 int32_t n_prompt_tokens_cache;353 int32_t n_tokens_cached;354 bool has_new_line;355 std::string stopping_word;356 stop_type stop = STOP_TYPE_NONE;357 358 bool post_sampling_probs;359 std::vector<completion_token_output> probs_output;360 std::vector<std::string> response_fields;361 362 task_params generation_params;363 364 // response formatting365 bool verbose = false;366 task_response_type res_type = TASK_RESPONSE_TYPE_NONE;367 std::string oaicompat_model;368 std::string oaicompat_cmpl_id;369 common_chat_msg oaicompat_msg; // to be populated by update()370 371 std::vector<common_chat_msg_diff> oaicompat_msg_diffs; // to be populated by update()372 bool is_updated = false;373 374 // for OpenAI Responses API375 std::string oai_resp_id;376 std::string oai_resp_reasoning_id;377 std::string oai_resp_message_id;378 379 virtual bool is_stop() override {380 return true; // in stream mode, final responses are considered stop381 }382 383 virtual json to_json() override;384 385 virtual void update(task_result_state & state) override {386 is_updated = true;387 oaicompat_msg = state.update_chat_msg(content, false, oaicompat_msg_diffs);388 389 oai_resp_id = state.oai_resp_id;390 oai_resp_reasoning_id = state.oai_resp_reasoning_id;391 oai_resp_message_id = state.oai_resp_message_id;392 }393 394 json to_json_non_oaicompat();395 396 json usage_json_oaicompat();397 398 json to_json_oaicompat();399 400 json to_json_oaicompat_chat();401 402 json to_json_oaicompat_chat_stream();403 404 json to_json_oaicompat_resp();405 406 json to_json_oaicompat_resp_stream();407 408 json to_json_oaicompat_asr();409 410 json to_json_anthropic();411 412 json to_json_anthropic_stream();413};414 415struct server_task_result_cmpl_partial : server_task_result {416 std::string content;417 llama_tokens tokens;418 419 int32_t n_decoded;420 int32_t n_prompt_tokens;421 int32_t n_prompt_tokens_cache;422 423 bool post_sampling_probs;424 bool is_progress = false;425 bool is_begin = false; // whether to send 200 status to HTTP client (begin of SSE stream)426 // ref: https://github.com/ggml-org/llama.cpp/pull/23884427 completion_token_output prob_output;428 result_timings timings;429 result_prompt_progress progress;430 431 // response formatting432 bool verbose = false;433 task_response_type res_type = TASK_RESPONSE_TYPE_NONE;434 std::string oaicompat_model;435 std::string oaicompat_cmpl_id;436 std::vector<common_chat_msg_diff> oaicompat_msg_diffs; // to be populated by update()437 bool is_updated = false;438 439 // Streaming state copied from task_result_state for this chunk440 bool thinking_block_started = false;441 bool text_block_started = false;442 443 // for OpenAI Responses API444 bool oai_resp_created = false;445 std::string oai_resp_id;446 std::string oai_resp_reasoning_id;447 std::string oai_resp_message_id;448 std::string oai_resp_fc_id;449 450 // for Anthropic API: track if any reasoning content has been generated451 bool anthropic_has_reasoning = false;452 453 virtual bool is_stop() override {454 return false; // in stream mode, partial responses are not considered stop455 }456 457 virtual void update(task_result_state & state) override;458 459 virtual json to_json() override;460 461 json to_json_non_oaicompat();462 463 json to_json_oaicompat();464 465 json to_json_oaicompat_chat();466 467 json to_json_oaicompat_resp();468 469 json to_json_oaicompat_asr();470 471 json to_json_anthropic();472};473 474struct server_task_result_embd : server_task_result {475 std::vector<std::vector<float>> embedding;476 477 int32_t n_tokens;478 479 // response formatting480 task_response_type res_type = TASK_RESPONSE_TYPE_NONE;481 482 virtual json to_json() override;483 484 json to_json_non_oaicompat();485 486 json to_json_oaicompat();487};488 489struct server_task_result_rerank : server_task_result {490 float score = -1e6;491 492 int32_t n_tokens;493 494 virtual json to_json() override;495};496 497struct server_task_result_error : server_task_result {498 error_type err_type = ERROR_TYPE_SERVER;499 std::string err_msg;500 501 // for ERROR_TYPE_EXCEED_CONTEXT_SIZE502 int32_t n_prompt_tokens = 0;503 int32_t n_ctx = 0;504 505 virtual bool is_error() override {506 return true;507 }508 509 virtual json to_json() override;510};511 512struct server_task_result_metrics : server_task_result {513 int n_idle_slots;514 int n_processing_slots;515 int n_tasks_deferred;516 int64_t t_start;517 518 // TODO: somehow reuse server_metrics in the future, instead of duplicating the fields519 uint64_t n_prompt_tokens_processed_total = 0;520 uint64_t t_prompt_processing_total = 0;521 uint64_t n_tokens_predicted_total = 0;522 uint64_t t_tokens_generation_total = 0;523 524 uint64_t n_tokens_max = 0;525 526 uint64_t n_prompt_tokens_processed = 0;527 uint64_t t_prompt_processing = 0;528 529 uint64_t n_tokens_predicted = 0;530 uint64_t t_tokens_generation = 0;531 532 uint64_t n_decode_total = 0;533 uint64_t n_busy_slots_total = 0;534 535 uint64_t n_draft_tokens_total = 0;536 uint64_t n_draft_accepted_total = 0;537 uint64_t n_draft_verif_steps_total = 0;538 std::vector<uint64_t> n_accepted_per_pos_total;539 540 // while we can also use std::vector<server_slot> this requires copying the slot object which can be quite messy541 // therefore, we use json to temporarily store the slot.to_json() result542 json slots_data = json::array();543 544 virtual json to_json() override;545};546 547struct server_task_result_slot_save_load : server_task_result {548 std::string filename;549 bool is_save; // true = save, false = load550 551 size_t n_tokens;552 size_t n_bytes;553 double t_ms;554 555 virtual json to_json() override;556};557 558struct server_task_result_slot_erase : server_task_result {559 size_t n_erased;560 561 virtual json to_json() override;562};563 564struct server_task_result_control : server_task_result {565 bool success = false;566 std::string message; // optional detail when success is false567 568 virtual json to_json() override {569 json out = json { { "success", success } };570 if (!message.empty()) {571 out["message"] = message;572 }573 return out;574 }575};576 577struct server_task_result_get_lora : server_task_result {578 struct lora {579 common_adapter_lora_info info;580 std::string alora_invocation_string;581 llama_tokens alora_invocation_tokens;582 };583 std::vector<lora> loras;584 585 virtual json to_json() override;586};587 588struct server_task_result_apply_lora : server_task_result {589 virtual json to_json() override;590};591 592struct server_prompt {593 server_tokens tokens;594 595 std::list<common_prompt_checkpoint> checkpoints;596 597 void clear() {598 tokens.clear();599 checkpoints.clear();600 }601 602 int n_tokens() const {603 return tokens.size();604 }605 606 server_prompt clone() const {607 return server_prompt {608 tokens.clone(),609 checkpoints,610 };611 }612};613 614struct server_prompt_data {615 std::vector<uint8_t> main;616 std::vector<uint8_t> drft;617 618 size_t size() const {619 return main.size() + drft.size();620 }621};622 623struct server_prompt_cache_state {624 server_prompt prompt;625 server_prompt_data data;626 627 size_t size() const {628 size_t res = data.size();629 630 for (const auto & ckpt : prompt.checkpoints) {631 res += ckpt.size();632 }633 634 return res;635 }636};637 638struct server_prompt_cache {639 server_prompt_cache(int32_t limit_size_mib, size_t limit_tokens) {640 this->limit_size = 1024ull*1024ull*(limit_size_mib < 0 ? 0 : limit_size_mib);641 this->limit_tokens = limit_tokens;642 }643 644 std::list<server_prompt_cache_state> states;645 646 // in bytes, 0 = no limit647 size_t limit_size = 0;648 649 // in tokens, 0 = no limit650 size_t limit_tokens = 0;651 652 size_t size() const;653 654 size_t n_tokens() const;655 656 server_prompt_cache_state * alloc(const server_prompt & prompt, size_t state_size_main, size_t state_size_drft);657 658 bool load(server_prompt & prompt, const server_tokens & tokens_new, llama_context * ctx_tgt, llama_context * ctx_dft, int32_t id_slot);659 660 void update();661};662 663// used exclusively by router mode664struct server_task_result_router : server_task_result {665 json data;666 virtual json to_json() override { return data; }667 virtual server_task_result * clone() const override {668 return new server_task_result_router(*this);669 }670};671 