Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
server-task.h671 linesDownload Raw Back to server
1#pragma once2 3#include "common.h"4#include "llama.h"5 6#include <string>7#include <unordered_set>8#include <list>9#include <map>10 11// TODO: prevent including the whole server-common.h as we only use server_tokens12#include "server-common.h"13 14using json = nlohmann::ordered_json;15 16enum server_task_type {17    SERVER_TASK_TYPE_COMPLETION,18    SERVER_TASK_TYPE_EMBEDDING,19    SERVER_TASK_TYPE_RERANK,20    SERVER_TASK_TYPE_INFILL,21    SERVER_TASK_TYPE_CANCEL,22    SERVER_TASK_TYPE_CONTROL,23    SERVER_TASK_TYPE_NEXT_RESPONSE,24    SERVER_TASK_TYPE_METRICS,25    SERVER_TASK_TYPE_SLOT_SAVE,26    SERVER_TASK_TYPE_SLOT_RESTORE,27    SERVER_TASK_TYPE_SLOT_ERASE,28    SERVER_TASK_TYPE_GET_LORA,29    SERVER_TASK_TYPE_SET_LORA,30};31 32// TODO: change this to more generic "response_format" to replace the "format_response_*" in server-common33enum task_response_type {34    TASK_RESPONSE_TYPE_NONE, // llama.cpp native format35    TASK_RESPONSE_TYPE_OAI_CHAT,36    TASK_RESPONSE_TYPE_OAI_CMPL,37    TASK_RESPONSE_TYPE_OAI_RESP,38    TASK_RESPONSE_TYPE_OAI_ASR, // transcriptions API39    TASK_RESPONSE_TYPE_OAI_EMBD,40    TASK_RESPONSE_TYPE_ANTHROPIC,41};42 43enum stop_type {44    STOP_TYPE_NONE,45    STOP_TYPE_EOS,46    STOP_TYPE_WORD,47    STOP_TYPE_LIMIT,48};49 50struct task_params {51    bool stream          = false;52    bool include_usage   = false;53    bool cache_prompt    = true; // remember the prompt to avoid reprocessing all prompt54    bool return_tokens   = false;55    bool return_progress = false;56 57    int32_t sse_ping_interval = 30; // seconds between SSE comment pings while the stream stays silent, -1 disables58 59    int32_t n_keep    =  0; // number of tokens to keep from initial prompt60    int32_t n_discard =  0; // number of tokens after n_keep that may be discarded when shifting context, 0 defaults to half61    int32_t n_predict = -1; // new tokens to predict62    int32_t n_indent  =  0; // minimum line indentation for the generated text in number of whitespace characters63    int32_t n_cmpl    =  1; // number of completions to generate from this prompt64 65    int32_t n_cache_reuse = 0; // min chunk size to attempt reusing from the cache via KV shifting (0 = disabled)66 67    int64_t t_max_prompt_ms  = -1; // TODO: implement68    int64_t t_max_predict_ms = -1; // if positive, limit the generation phase to this time limit69 70    std::map<int, float> lora; // mapping adapter ID -> scale71 72    std::vector<std::string> antiprompt;73    std::vector<std::string> response_fields;74 75    bool timings_per_token   = false;76    bool post_sampling_probs = false;77 78    struct common_params_sampling sampling;79    struct common_params_speculative speculative;80 81    // response formatting82    bool               verbose  = false;83    task_response_type res_type = TASK_RESPONSE_TYPE_NONE;84    std::string        oaicompat_model;85    std::string        oaicompat_cmpl_id;86 87    // realtime control (SERVER_TASK_TYPE_CONTROL)88    std::string        control_action;89    std::string        control_cmpl_id;90 91    // per-request parameters for chat parsing92    common_chat_parser_params chat_parser_params;93 94    // message spans for checkpointing95    common_chat_msg_spans message_spans;96 97    // Embeddings98    int32_t embd_normalize = 2; // (-1=none, 0=max absolute int16, 1=taxicab, 2=Euclidean/L2, >2=p-norm)99 100    json format_logit_bias(const std::vector<llama_logit_bias> & logit_bias) const;101    json to_json(bool only_metrics = false) const;102};103 104// struct for tracking the state of a task (e.g., for streaming)105struct task_result_state {106    // tracking diffs for partial tool calls107    std::vector<common_chat_msg_diff> diffs;108    common_chat_parser_params chat_parser_params;109    common_chat_msg chat_msg;110    std::string generated_text; // append new chunks of generated text here111    std::vector<std::string> generated_tool_call_ids;112    std::unordered_set<size_t> sent_tool_call_names;113 114    // for OpenAI Responses and Anthropic streaming API:115    // track output item / content block state across chunks116    bool thinking_block_started = false;117    bool text_block_started = false;118 119    // for OpenAI Responses streaming API120    bool oai_resp_created = false;121    const std::string oai_resp_id;122    const std::string oai_resp_reasoning_id;123    const std::string oai_resp_message_id;124    std::string oai_resp_fc_id; // function call ID for current args delta125 126    task_result_state(const common_chat_parser_params & chat_parser_params);127 128    // parse partial tool calls and update the internal state129    common_chat_msg update_chat_msg(130        const std::string & text_added,131        bool is_partial,132        std::vector<common_chat_msg_diff> & diffs,133        bool filter_tool_calls = false);134};135 136struct server_task {137    int id = -1; // to be filled by server_queue138 139    // TODO @ngxson : remove this field and implement a mapping task_id -> idx in the response_reader140    size_t index = 0; // used when there are multiple prompts (batch request)141 142    // used by SERVER_TASK_TYPE_CANCEL143    int id_target = -1;144    int id_slot   = -1;145 146    // used by parallel sampling (multiple completions from same prompt)147    int id_parent  = -1;148    // temporary store of child tasks for scheduling149    // note: accessing to elements is invalid after the task is moved to server_slot150    std::vector<server_task> child_tasks;151 152    // used by SERVER_TASK_TYPE_INFERENCE153    task_params   params;154    server_tokens tokens;155 156    // only used by CLI, this allow tokenizing CLI inputs on server side157    // we need this because mtmd_context and vocab are not accessible outside of server_context158    bool                    cli = false;159    std::string             cli_prompt;160    std::vector<raw_buffer> cli_files;161 162    server_task_type type;163 164    // used by SERVER_TASK_TYPE_SLOT_SAVE, SERVER_TASK_TYPE_SLOT_RESTORE, SERVER_TASK_TYPE_SLOT_ERASE165    struct slot_action {166        int id_slot;167        std::string filename;168        std::string filepath;169    };170    slot_action slot_action;171 172    // used by SERVER_TASK_TYPE_METRICS173    bool metrics_reset_bucket = false;174 175    // used by SERVER_TASK_TYPE_SET_LORA176    std::map<int, float> set_lora; // mapping adapter ID -> scale177 178    server_task() = default;179 180    server_task(server_task_type type) : type(type) {}181 182    int32_t n_tokens() const {183        return tokens.size();184    }185 186    bool need_embd() const {187        switch (type) {188            case SERVER_TASK_TYPE_EMBEDDING:189            case SERVER_TASK_TYPE_RERANK:190                return true;191            default:192                return false;193        }194    }195 196    bool need_logits() const {197        switch (type) {198            case SERVER_TASK_TYPE_COMPLETION:199            case SERVER_TASK_TYPE_INFILL:200                return true;201            default:202                return false;203        }204    }205 206    bool need_sampling() const {207        switch (type) {208            case SERVER_TASK_TYPE_COMPLETION:209            case SERVER_TASK_TYPE_INFILL:210                return true;211            default:212                return false;213        }214    }215 216    // utility function217    static std::unordered_set<int> get_list_id(const std::vector<server_task> & tasks) {218        std::unordered_set<int> ids(tasks.size());219        for (size_t i = 0; i < tasks.size(); i++) {220            ids.insert(tasks[i].id);221            for (auto & child : tasks[i].child_tasks) {222                ids.insert(child.id);223            }224        }225        return ids;226    }227 228    void add_child(int id_parent, int id_child) {229        server_task copy;230 231        copy.id        = id_child;232        copy.id_parent = id_parent;233        copy.params    = params;234        copy.type      = type;235        copy.tokens    = tokens.clone();236        copy.id_slot   = -1; // child tasks cannot specify slot237 238        // use different sampling seed for each child239        // note: https://github.com/ggml-org/llama.cpp/pull/18700#discussion_r2675115723240        if (copy.params.sampling.seed != LLAMA_DEFAULT_SEED) {241            copy.params.sampling.seed += (uint32_t)child_tasks.size() + 1;242        }243 244        child_tasks.push_back(std::move(copy));245    }246 247    // the task will be moved into queue, then onto slots248    // however, the state must be kept by caller (e.g., HTTP thread)249    task_result_state create_state() const {250        return task_result_state(params.chat_parser_params);251    }252 253    bool is_parent() const {254        return child_tasks.size() > 0;255    }256 257    bool is_child() const {258        return id_parent != -1;259    }260};261 262struct result_timings {263    int32_t cache_n = -1;264 265    int32_t prompt_n = -1;266    double prompt_ms = 0.0;267    double prompt_per_token_ms = 0.0;268    double prompt_per_second = 0.0;269 270    int32_t predicted_n = -1;271    double predicted_ms = 0.0;272    double predicted_per_token_ms = 0.0;273    double predicted_per_second = 0.0;274 275    // Optional speculative metrics - only included when > 0276    int32_t draft_n = 0;277    int32_t draft_n_accepted = 0;278 279    json to_json() const;280};281 282struct result_prompt_progress {283    int32_t total = 0;284    int32_t cache = 0;285    int32_t processed = 0;286    int64_t time_ms = 0;287 288    json to_json() const;289};290 291struct server_task_result {292    int id           = -1;293    int id_slot      = -1;294 295    // TODO @ngxson : remove this field and implement a mapping task_id -> idx in the response_reader296    size_t index = 0; // to be used for batched tasks297 298    virtual bool is_error() {299        // only used by server_task_result_error300        return false;301    }302    virtual bool is_stop() {303        // only used by server_task_result_cmpl_*304        return true;305    }306    virtual void update(task_result_state &) {307        // only used by server_task_result_cmpl_*308    }309    virtual json to_json() = 0;310    virtual ~server_task_result() = default;311    virtual server_task_result * clone() const {312        GGML_ABORT("not implemented for this task type");313    }314};315 316// using shared_ptr for polymorphism of server_task_result317using server_task_result_ptr = std::unique_ptr<server_task_result>;318 319struct completion_token_output {320    llama_token tok;321    float prob;322    std::string text_to_send;323    struct prob_info {324        llama_token tok;325        std::string txt;326        float prob;327    };328    std::vector<prob_info> probs;329 330    json to_json(bool post_sampling_probs) const;331 332    static json probs_vector_to_json(const std::vector<completion_token_output> & probs, bool post_sampling_probs);333 334    static float logarithm(float x);335 336    static std::vector<unsigned char> str_to_bytes(const std::string & str);337 338};339 340struct server_task_result_cmpl_final : server_task_result {341    std::string content;342    llama_tokens tokens;343 344    bool stream;345    bool include_usage;346    result_timings timings;347    std::string prompt;348 349    bool truncated;350    int32_t n_decoded;351    int32_t n_prompt_tokens;352    int32_t n_prompt_tokens_cache;353    int32_t n_tokens_cached;354    bool has_new_line;355    std::string stopping_word;356    stop_type stop = STOP_TYPE_NONE;357 358    bool post_sampling_probs;359    std::vector<completion_token_output> probs_output;360    std::vector<std::string>  response_fields;361 362    task_params generation_params;363 364    // response formatting365    bool               verbose  = false;366    task_response_type res_type = TASK_RESPONSE_TYPE_NONE;367    std::string        oaicompat_model;368    std::string        oaicompat_cmpl_id;369    common_chat_msg    oaicompat_msg; // to be populated by update()370 371    std::vector<common_chat_msg_diff> oaicompat_msg_diffs; // to be populated by update()372    bool is_updated = false;373 374    // for OpenAI Responses API375    std::string oai_resp_id;376    std::string oai_resp_reasoning_id;377    std::string oai_resp_message_id;378 379    virtual bool is_stop() override {380        return true; // in stream mode, final responses are considered stop381    }382 383    virtual json to_json() override;384 385    virtual void update(task_result_state & state) override {386        is_updated = true;387        oaicompat_msg = state.update_chat_msg(content, false, oaicompat_msg_diffs);388 389        oai_resp_id = state.oai_resp_id;390        oai_resp_reasoning_id = state.oai_resp_reasoning_id;391        oai_resp_message_id = state.oai_resp_message_id;392    }393 394    json to_json_non_oaicompat();395 396    json usage_json_oaicompat();397 398    json to_json_oaicompat();399 400    json to_json_oaicompat_chat();401 402    json to_json_oaicompat_chat_stream();403 404    json to_json_oaicompat_resp();405 406    json to_json_oaicompat_resp_stream();407 408    json to_json_oaicompat_asr();409 410    json to_json_anthropic();411 412    json to_json_anthropic_stream();413};414 415struct server_task_result_cmpl_partial : server_task_result {416    std::string  content;417    llama_tokens tokens;418 419    int32_t n_decoded;420    int32_t n_prompt_tokens;421    int32_t n_prompt_tokens_cache;422 423    bool post_sampling_probs;424    bool is_progress = false;425    bool is_begin = false; // whether to send 200 status to HTTP client (begin of SSE stream)426                           // ref: https://github.com/ggml-org/llama.cpp/pull/23884427    completion_token_output prob_output;428    result_timings timings;429    result_prompt_progress progress;430 431    // response formatting432    bool               verbose  = false;433    task_response_type res_type = TASK_RESPONSE_TYPE_NONE;434    std::string        oaicompat_model;435    std::string        oaicompat_cmpl_id;436    std::vector<common_chat_msg_diff> oaicompat_msg_diffs; // to be populated by update()437    bool is_updated = false;438 439    // Streaming state copied from task_result_state for this chunk440    bool thinking_block_started = false;441    bool text_block_started     = false;442 443    // for OpenAI Responses API444    bool oai_resp_created = false;445    std::string oai_resp_id;446    std::string oai_resp_reasoning_id;447    std::string oai_resp_message_id;448    std::string oai_resp_fc_id;449 450    // for Anthropic API: track if any reasoning content has been generated451    bool anthropic_has_reasoning = false;452 453    virtual bool is_stop() override {454        return false; // in stream mode, partial responses are not considered stop455    }456 457    virtual void update(task_result_state & state) override;458 459    virtual json to_json() override;460 461    json to_json_non_oaicompat();462 463    json to_json_oaicompat();464 465    json to_json_oaicompat_chat();466 467    json to_json_oaicompat_resp();468 469    json to_json_oaicompat_asr();470 471    json to_json_anthropic();472};473 474struct server_task_result_embd : server_task_result {475    std::vector<std::vector<float>> embedding;476 477    int32_t n_tokens;478 479    // response formatting480    task_response_type res_type = TASK_RESPONSE_TYPE_NONE;481 482    virtual json to_json() override;483 484    json to_json_non_oaicompat();485 486    json to_json_oaicompat();487};488 489struct server_task_result_rerank : server_task_result {490    float score = -1e6;491 492    int32_t n_tokens;493 494    virtual json to_json() override;495};496 497struct server_task_result_error : server_task_result {498    error_type err_type = ERROR_TYPE_SERVER;499    std::string err_msg;500 501    // for ERROR_TYPE_EXCEED_CONTEXT_SIZE502    int32_t n_prompt_tokens = 0;503    int32_t n_ctx           = 0;504 505    virtual bool is_error() override {506        return true;507    }508 509    virtual json to_json() override;510};511 512struct server_task_result_metrics : server_task_result {513    int n_idle_slots;514    int n_processing_slots;515    int n_tasks_deferred;516    int64_t t_start;517 518    // TODO: somehow reuse server_metrics in the future, instead of duplicating the fields519    uint64_t n_prompt_tokens_processed_total = 0;520    uint64_t t_prompt_processing_total       = 0;521    uint64_t n_tokens_predicted_total        = 0;522    uint64_t t_tokens_generation_total       = 0;523 524    uint64_t n_tokens_max = 0;525 526    uint64_t n_prompt_tokens_processed = 0;527    uint64_t t_prompt_processing       = 0;528 529    uint64_t n_tokens_predicted  = 0;530    uint64_t t_tokens_generation = 0;531 532    uint64_t n_decode_total     = 0;533    uint64_t n_busy_slots_total = 0;534 535    uint64_t n_draft_tokens_total      = 0;536    uint64_t n_draft_accepted_total    = 0;537    uint64_t n_draft_verif_steps_total = 0;538    std::vector<uint64_t> n_accepted_per_pos_total;539 540    // while we can also use std::vector<server_slot> this requires copying the slot object which can be quite messy541    // therefore, we use json to temporarily store the slot.to_json() result542    json slots_data = json::array();543 544    virtual json to_json() override;545};546 547struct server_task_result_slot_save_load : server_task_result {548    std::string filename;549    bool is_save; // true = save, false = load550 551    size_t n_tokens;552    size_t n_bytes;553    double t_ms;554 555    virtual json to_json() override;556};557 558struct server_task_result_slot_erase : server_task_result {559    size_t n_erased;560 561    virtual json to_json() override;562};563 564struct server_task_result_control : server_task_result {565    bool        success = false;566    std::string message; // optional detail when success is false567 568    virtual json to_json() override {569        json out = json { { "success", success } };570        if (!message.empty()) {571            out["message"] = message;572        }573        return out;574    }575};576 577struct server_task_result_get_lora : server_task_result {578    struct lora {579        common_adapter_lora_info info;580        std::string  alora_invocation_string;581        llama_tokens alora_invocation_tokens;582    };583    std::vector<lora> loras;584 585    virtual json to_json() override;586};587 588struct server_task_result_apply_lora : server_task_result {589    virtual json to_json() override;590};591 592struct server_prompt {593    server_tokens tokens;594 595    std::list<common_prompt_checkpoint> checkpoints;596 597    void clear() {598        tokens.clear();599        checkpoints.clear();600    }601 602    int n_tokens() const {603        return tokens.size();604    }605 606    server_prompt clone() const {607        return server_prompt {608            tokens.clone(),609            checkpoints,610        };611    }612};613 614struct server_prompt_data {615    std::vector<uint8_t> main;616    std::vector<uint8_t> drft;617 618    size_t size() const {619        return main.size() + drft.size();620    }621};622 623struct server_prompt_cache_state {624    server_prompt prompt;625    server_prompt_data data;626 627    size_t size() const {628        size_t res = data.size();629 630        for (const auto & ckpt : prompt.checkpoints) {631            res += ckpt.size();632        }633 634        return res;635    }636};637 638struct server_prompt_cache {639    server_prompt_cache(int32_t limit_size_mib, size_t limit_tokens) {640        this->limit_size   = 1024ull*1024ull*(limit_size_mib < 0 ? 0 : limit_size_mib);641        this->limit_tokens = limit_tokens;642    }643 644    std::list<server_prompt_cache_state> states;645 646    // in bytes, 0 = no limit647    size_t limit_size = 0;648 649    // in tokens, 0 = no limit650    size_t limit_tokens = 0;651 652    size_t size() const;653 654    size_t n_tokens() const;655 656    server_prompt_cache_state * alloc(const server_prompt & prompt, size_t state_size_main, size_t state_size_drft);657 658    bool load(server_prompt & prompt, const server_tokens & tokens_new, llama_context * ctx_tgt, llama_context * ctx_dft, int32_t id_slot);659 660    void update();661};662 663// used exclusively by router mode664struct server_task_result_router : server_task_result {665    json data;666    virtual json to_json() override { return data; }667    virtual server_task_result * clone() const override {668        return new server_task_result_router(*this);669    }670};671 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai