Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
server-common.h449 linesDownload Raw Back to server
1#pragma once2 3#include "common.h"4#include "log.h"5#include "llama.h"6#include "chat.h"7#include "mtmd.h"8 9#define JSON_ASSERT GGML_ASSERT10#include <nlohmann/json.hpp>11 12#include <atomic>13#include <chrono>14#include <condition_variable>15#include <cinttypes>16#include <functional>17#include <mutex>18#include <queue>19#include <string>20#include <vector>21 22using json = nlohmann::ordered_json;23 24#define SLT_DBG(slot, fmt, ...) LOG_DBG("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)25#define SLT_TRC(slot, fmt, ...) LOG_TRC("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)26#define SLT_INF(slot, fmt, ...) LOG_INF("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)27#define SLT_WRN(slot, fmt, ...) LOG_WRN("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)28#define SLT_ERR(slot, fmt, ...) LOG_ERR("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)29#define SLT_CNT(slot, fmt, ...) LOG_CNT(""                                 fmt,                                                                __VA_ARGS__)30 31#define SRV_DBG(fmt, ...) LOG_DBG("srv  %12.*s: " fmt, 12, __func__, __VA_ARGS__)32#define SRV_TRC(fmt, ...) LOG_TRC("srv  %12.*s: " fmt, 12, __func__, __VA_ARGS__)33#define SRV_INF(fmt, ...) LOG_INF("srv  %12.*s: " fmt, 12, __func__, __VA_ARGS__)34#define SRV_WRN(fmt, ...) LOG_WRN("srv  %12.*s: " fmt, 12, __func__, __VA_ARGS__)35#define SRV_ERR(fmt, ...) LOG_ERR("srv  %12.*s: " fmt, 12, __func__, __VA_ARGS__)36#define SRV_CNT(fmt, ...) LOG_CNT(""              fmt,               __VA_ARGS__)37 38using raw_buffer = std::vector<uint8_t>;39 40template <typename T>41static T json_value(const json & body, const std::string & key, const T & default_value) {42    // Fallback null to default value43    if (body.contains(key) && !body.at(key).is_null()) {44        try {45            return body.at(key);46        } catch (NLOHMANN_JSON_NAMESPACE::detail::type_error const & err) {47            LOG_WRN("Wrong type supplied for parameter '%s'. Expected '%s', using default value: %s\n", key.c_str(), json(default_value).type_name(), err.what());48            return default_value;49        }50    } else {51        return default_value;52    }53}54 55// https://community.openai.com/t/openai-chat-list-of-error-codes-and-types/357791/1156enum error_type {57    ERROR_TYPE_INVALID_REQUEST,58    ERROR_TYPE_AUTHENTICATION,59    ERROR_TYPE_SERVER,60    ERROR_TYPE_NOT_FOUND,61    ERROR_TYPE_PERMISSION,62    ERROR_TYPE_UNAVAILABLE, // custom error63    ERROR_TYPE_NOT_SUPPORTED, // custom error64    ERROR_TYPE_EXCEED_CONTEXT_SIZE, // custom error65};66 67// thin wrapper around common_grammar_trigger with (de)serialization functions68struct server_grammar_trigger {69    common_grammar_trigger value;70 71    server_grammar_trigger() = default;72    server_grammar_trigger(const common_grammar_trigger & value) : value(value) {}73    server_grammar_trigger(const json & in) {74        value.type = (common_grammar_trigger_type) in.at("type").get<int>();75        value.value = in.at("value").get<std::string>();76        if (value.type == COMMON_GRAMMAR_TRIGGER_TYPE_TOKEN) {77            value.token = (llama_token) in.at("token").get<int>();78        }79    }80 81    json to_json() const {82        json out {83            {"type", (int) value.type},84            {"value", value.value},85        };86        if (value.type == COMMON_GRAMMAR_TRIGGER_TYPE_TOKEN) {87            out["token"] = (int) value.token;88        }89        return out;90    }91};92 93json format_error_response(const std::string & message, const enum error_type type);94 95//96// random string / id97//98 99std::string random_string();100std::string gen_chatcmplid();101std::string gen_tool_call_id();102 103// get a random marker; note: each time the server restarts, the marker will be different104const char * get_media_marker();105 106//107// lora utils108//109 110// check whether the given lora set has only aloras activated (empty => false)111bool lora_all_alora(const std::vector<common_adapter_lora_info> & loras);112 113// if the two sets of loras are different, they require a cache clear unless the114// change is only from aloras to aloras.115bool lora_should_clear_cache(116        const std::vector<common_adapter_lora_info> & current,117        const std::vector<common_adapter_lora_info> & next);118 119std::map<int, float> parse_lora_request(const json & data);120 121bool are_lora_equal(122        const std::vector<common_adapter_lora_info> & l1,123        const std::vector<common_adapter_lora_info> & l2);124 125// get the ids of all enabled loras126std::vector<size_t> lora_get_enabled_ids(const std::vector<common_adapter_lora_info> & loras);127 128//129// server_tokens130//131 132/**133 * server_tokens is a helper to manage the input tokens and image for the server.134 * it is made this way to simplify the logic of KV cache management.135 */136struct server_tokens {137    bool has_mtmd = false;138 139private: // disallow accessing these members directly, risking out-of-sync140 141    // map a **start** index in tokens to the image chunk142    // note: the order need to be in-sync with tokens143    std::map<size_t, mtmd::input_chunk_ptr> map_idx_to_media;144 145    // list of tokens146    //   if the token is LLAMA_TOKEN_NULL, it indicates that this position is occupied by media chunk147    //   otherwise, it is a normal text token148    // note: a non-text chunk can occupy multiple tokens (aka memory cells) in the token list149    // note(2): for M-RoPE, an image can occupy different number of pos; do not assume 1-to-1 mapping tokens <-> pos150    llama_tokens tokens;151 152    // for ex. with input of 5 text tokens and 2 images (each image occupies 3 tokens and 2 pos):153    //      [0] [1] [2] [3] [4] [img0] [img0] [img0] [img1] [img1] [img1]154    // idx  0   1   2   3   4   5      6      7      8      9      10155    // pos  0   1   2   3   4   5      5      5      7      7      7156    // map_idx_to_media will contain: {5, img0}, {8, img1}157 158public:159    server_tokens() = default;160    ~server_tokens() = default;161 162    // Prevent copying163    // TODO: server_tokens should be copyable - remove this:164    server_tokens(const server_tokens&) = delete;165    server_tokens& operator=(const server_tokens&) = delete;166 167    // Allow moving (usually implicitly generated if members are movable)168    server_tokens(server_tokens&&) = default;169    server_tokens& operator=(server_tokens&&) = default;170 171    // Allow accessing elements using [] operator172    llama_token operator[](size_t index) { return tokens[index]; }173    const llama_token& operator[](size_t index) const { return tokens[index]; }174 175    server_tokens(mtmd::input_chunks & mtmd_chunks, bool has_mtmd);176    server_tokens(const llama_tokens & tokens, bool has_mtmd);177 178    // for debugging179    std::string str() const;180 181    // the next position after n_tokens. if n_tokens < 0, return the next position after all tokens.182    llama_pos pos_next(int64_t n_tokens = -1) const;183 184    // number of tokens with position < max_pos185    size_t size_up_to_pos(llama_pos max_pos) const;186 187    const mtmd::input_chunk_ptr & find_chunk(size_t idx) const;188 189    // find next media chunk after idx190    // returns a pair of pointer to the chunk (nullptr if not found) and its start index in tokens191    std::pair<const mtmd::input_chunk_ptr *, size_t> find_next_media_chunk(size_t idx) const;192 193    void push_back(llama_token tok);194 195    // will create a copy of the chunk if it contains non-text data196    void push_back(const mtmd_input_chunk * chunk);197 198    // appends server tokens, updates the media map. copies media chunks.199    void push_back(server_tokens & tokens);200 201    // for compatibility with context shift and prompt truncation202    void insert(const llama_tokens & inp_tokens);203 204    // for compatibility with speculative decoding, ctx shift, slot save/load205    const llama_tokens & get_tokens() const;206 207    llama_tokens get_text_tokens() const;208 209    // for compatibility with speculative decoding210    void set_token(llama_pos pos, llama_token id);211 212    size_t size() const { return tokens.size(); }213 214    bool empty() const { return tokens.empty(); }215 216    // true if the sequence actually contains image/audio chunks.217    bool has_media() const { return !map_idx_to_media.empty(); }218 219    void clear() {220        map_idx_to_media.clear();221        tokens.clear();222    }223 224    void keep_first(size_t n);225 226    std::string detokenize(const llama_context * ctx, bool special) const;227 228    size_t get_common_prefix(const server_tokens & b) const;229 230    // split the tokens into message spans, skipping over media chunks231    common_chat_msg_spans find_message_spans(const common_chat_msg_delimiters & delims) const;232 233    // make sure all text tokens are within the vocab range234    bool validate(const struct llama_context * ctx) const;235 236    server_tokens clone() const;237};238 239 240//241// tokenizer and input processing utils242//243 244bool json_is_array_of_numbers(const json & data);245 246// is array having BOTH numbers & strings?247bool json_is_array_of_mixed_numbers_strings(const json & data);248 249// does array have any individual integers/tokens?250bool json_is_array_and_contains_numbers(const json & data);251 252// get value by path(key1 / key2)253json json_get_nested_values(const std::vector<std::string> & paths, const json & js);254 255/**256 * this handles 2 cases:257 * - only string, example: "string"258 * - mixed string and tokens, example: [12, 34, "string", 56, 78]259 */260llama_tokens tokenize_mixed(const llama_vocab * vocab, const json & json_prompt, bool add_special, bool parse_special);261 262// return the last index of character that can form a valid string263// if the last character is potentially cut in half, return the index before the cut264// if validate_utf8(text) == text.size(), then the whole text is valid utf8265size_t validate_utf8(const std::string& text);266 267// process mtmd prompt, return the server_tokens containing both text tokens and media chunks268// if is_placeholder is true, the media chunk will be treated as placeholder for counting tokens; the output tokens are not usable for actual inference (e.g. for submitting a task to server_queue)269server_tokens process_mtmd_prompt(mtmd_context * mctx, const std::string & prompt, const std::vector<raw_buffer> & files, bool is_placeholder = false);270 271/**272 * break the input "prompt" object into multiple prompt if needed, then tokenize them273 * this supports these cases:274 * - "prompt": "string"275 * - "prompt": [12, 34, 56]276 * - "prompt": [12, 34, "string", 56, 78]277 * - "prompt": { "prompt_string": "string", "multimodal_data": [ "base64" ] }278 * and multiple prompts (multi-tasks):279 * - "prompt": ["string1", "string2"]280 * - "prompt": ["string1", [12, 34, 56]]281 * - "prompt": [[12, 34, 56], [78, 90, 12]]282 * - "prompt": [[12, 34, "string", 56, 78], [12, 34, 56], { "prompt_string": "string", "multimodal_data": [ "base64" ]}]283 */284std::vector<server_tokens> tokenize_input_prompts(285                                        const llama_vocab * vocab,286                                        mtmd_context * mctx,287                                        const json & json_prompt,288                                        bool add_special,289                                        bool parse_special);290 291//292// OAI utils293//294 295// global server parameters for chat formatting / parsing296struct server_chat_params {297    bool use_jinja;298    bool prefill_assistant;299    common_reasoning_format reasoning_format;300    std::map<std::string, std::string> chat_template_kwargs; // mapping key --> json value301    common_chat_templates_ptr tmpls;302    bool allow_image;303    bool allow_audio;304    bool allow_video;305    bool enable_thinking = true;306    int  reasoning_budget = -1;307    std::string reasoning_budget_message;308    std::string media_path;309    bool force_pure_content = false;310};311 312// used by /completions endpoint313json oaicompat_completion_params_parse(const json & body);314 315// used by /chat/completions endpoint316json oaicompat_chat_params_parse(317    json & body, /* openai api json semantics */318    const server_chat_params & opt,319    std::vector<raw_buffer> & out_files);320 321// TODO: move it to server-task.cpp322json format_embeddings_response_oaicompat(323    const json & request,324    const std::string & model_name,325    const json & embeddings,326    bool use_base64 = false);327 328// TODO: move it to server-task.cpp329json format_response_rerank(330        const json & request,331        const std::string & model_name,332        const json & ranks,333        bool is_tei_format,334        std::vector<std::string> & texts,335        int top_n);336 337//338// other utils339//340 341std::vector<llama_token_data> get_token_probabilities(llama_context * ctx, int idx, size_t n_top);342 343std::string safe_json_to_str(const json & data);344 345std::string tokens_to_str(llama_context * ctx, const llama_tokens & tokens);346std::string tokens_to_str(const llama_vocab * vocab, const llama_tokens & tokens);347 348// format incomplete utf-8 multibyte character for output349std::string tokens_to_output_formatted_string(const llama_context * ctx, const llama_token token);350 351// format server-sent event (SSE), return the formatted string to send352// note: if data is a json array, it will be sent as multiple events, one per item353std::string format_oai_sse(const json & data);354 355std::string format_oai_resp_sse(const json & data);356 357// format Anthropic-style SSE with event types358std::string format_anthropic_sse(const json & data);359 360bool is_valid_utf8(const std::string & str);361 362//363// formatting output responses364// TODO: move these to server-task.cpp365//366 367llama_tokens format_prompt_infill(368        const llama_vocab * vocab,369        const json & input_prefix,370        const json & input_suffix,371        const json & input_extra,372        const int n_batch,373        const int n_predict,374        const int n_ctx,375        const bool spm_infill,376        const llama_tokens & tokens_prompt);377 378// format rerank task: [BOS]query[EOS][SEP]doc[EOS].379server_tokens format_prompt_rerank(380        const struct llama_model * model,381        const struct llama_vocab * vocab,382        mtmd_context * mctx,383        const std::string & query,384        const std::string & doc);385 386// simple implementation of a pipe387// used for streaming data between threads388template<typename T>389struct server_pipe {390    std::mutex mutex;391    std::condition_variable cv;392    std::queue<T> queue;393    std::atomic<bool> writer_closed{false};394    std::atomic<bool> reader_closed{false};395 396    // 0 = unbounded (default)397    // > 0, write() drops the oldest item once the queue is full398    size_t max_size = 0;399 400    void close_write() {401        writer_closed.store(true, std::memory_order_relaxed);402        cv.notify_all();403    }404 405    void close_read() {406        reader_closed.store(true, std::memory_order_relaxed);407        cv.notify_all();408    }409 410    // close_on_stop = true: should_stop means the reader is gone for good, so the writer is told the pipe is broken.411    // close_on_stop = false: should_stop is a per-read deadline and further reads still come, so the pipe stays usable.412    bool read(T & output, const std::function<bool()> & should_stop, bool close_on_stop = true) {413        std::unique_lock<std::mutex> lk(mutex);414        constexpr auto poll_interval = std::chrono::milliseconds(500);415        while (true) {416            if (!queue.empty()) {417                output = std::move(queue.front());418                queue.pop();419                return true;420            }421            if (writer_closed.load()) {422                return false; // clean EOF423            }424            if (should_stop && should_stop()) { // a null should_stop means "never stop"425                if (close_on_stop) {426                    close_read(); // signal broken pipe to writer427                }428                return false; // cancelled / deadline reached429            }430            cv.wait_for(lk, poll_interval);431        }432    }433 434    bool write(T && data) {435        std::lock_guard<std::mutex> lk(mutex);436        if (reader_closed.load()) {437            return false; // broken pipe438        }439        if (max_size > 0) {440            while (queue.size() >= max_size) {441                queue.pop(); // drop oldest to stay bounded442            }443        }444        queue.push(std::move(data));445        cv.notify_one();446        return true;447    }448};449 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai