Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#pragma once2 3#include "common.h"4#include "log.h"5#include "llama.h"6#include "chat.h"7#include "mtmd.h"8 9#define JSON_ASSERT GGML_ASSERT10#include <nlohmann/json.hpp>11 12#include <atomic>13#include <chrono>14#include <condition_variable>15#include <cinttypes>16#include <functional>17#include <mutex>18#include <queue>19#include <string>20#include <vector>21 22using json = nlohmann::ordered_json;23 24#define SLT_DBG(slot, fmt, ...) LOG_DBG("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)25#define SLT_TRC(slot, fmt, ...) LOG_TRC("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)26#define SLT_INF(slot, fmt, ...) LOG_INF("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)27#define SLT_WRN(slot, fmt, ...) LOG_WRN("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)28#define SLT_ERR(slot, fmt, ...) LOG_ERR("slot %12.*s: id %2d | task %d | " fmt, 12, __func__, (slot).id, ((slot).task ? (slot).task->id : -1), __VA_ARGS__)29#define SLT_CNT(slot, fmt, ...) LOG_CNT("" fmt, __VA_ARGS__)30 31#define SRV_DBG(fmt, ...) LOG_DBG("srv %12.*s: " fmt, 12, __func__, __VA_ARGS__)32#define SRV_TRC(fmt, ...) LOG_TRC("srv %12.*s: " fmt, 12, __func__, __VA_ARGS__)33#define SRV_INF(fmt, ...) LOG_INF("srv %12.*s: " fmt, 12, __func__, __VA_ARGS__)34#define SRV_WRN(fmt, ...) LOG_WRN("srv %12.*s: " fmt, 12, __func__, __VA_ARGS__)35#define SRV_ERR(fmt, ...) LOG_ERR("srv %12.*s: " fmt, 12, __func__, __VA_ARGS__)36#define SRV_CNT(fmt, ...) LOG_CNT("" fmt, __VA_ARGS__)37 38using raw_buffer = std::vector<uint8_t>;39 40template <typename T>41static T json_value(const json & body, const std::string & key, const T & default_value) {42 // Fallback null to default value43 if (body.contains(key) && !body.at(key).is_null()) {44 try {45 return body.at(key);46 } catch (NLOHMANN_JSON_NAMESPACE::detail::type_error const & err) {47 LOG_WRN("Wrong type supplied for parameter '%s'. Expected '%s', using default value: %s\n", key.c_str(), json(default_value).type_name(), err.what());48 return default_value;49 }50 } else {51 return default_value;52 }53}54 55// https://community.openai.com/t/openai-chat-list-of-error-codes-and-types/357791/1156enum error_type {57 ERROR_TYPE_INVALID_REQUEST,58 ERROR_TYPE_AUTHENTICATION,59 ERROR_TYPE_SERVER,60 ERROR_TYPE_NOT_FOUND,61 ERROR_TYPE_PERMISSION,62 ERROR_TYPE_UNAVAILABLE, // custom error63 ERROR_TYPE_NOT_SUPPORTED, // custom error64 ERROR_TYPE_EXCEED_CONTEXT_SIZE, // custom error65};66 67// thin wrapper around common_grammar_trigger with (de)serialization functions68struct server_grammar_trigger {69 common_grammar_trigger value;70 71 server_grammar_trigger() = default;72 server_grammar_trigger(const common_grammar_trigger & value) : value(value) {}73 server_grammar_trigger(const json & in) {74 value.type = (common_grammar_trigger_type) in.at("type").get<int>();75 value.value = in.at("value").get<std::string>();76 if (value.type == COMMON_GRAMMAR_TRIGGER_TYPE_TOKEN) {77 value.token = (llama_token) in.at("token").get<int>();78 }79 }80 81 json to_json() const {82 json out {83 {"type", (int) value.type},84 {"value", value.value},85 };86 if (value.type == COMMON_GRAMMAR_TRIGGER_TYPE_TOKEN) {87 out["token"] = (int) value.token;88 }89 return out;90 }91};92 93json format_error_response(const std::string & message, const enum error_type type);94 95//96// random string / id97//98 99std::string random_string();100std::string gen_chatcmplid();101std::string gen_tool_call_id();102 103// get a random marker; note: each time the server restarts, the marker will be different104const char * get_media_marker();105 106//107// lora utils108//109 110// check whether the given lora set has only aloras activated (empty => false)111bool lora_all_alora(const std::vector<common_adapter_lora_info> & loras);112 113// if the two sets of loras are different, they require a cache clear unless the114// change is only from aloras to aloras.115bool lora_should_clear_cache(116 const std::vector<common_adapter_lora_info> & current,117 const std::vector<common_adapter_lora_info> & next);118 119std::map<int, float> parse_lora_request(const json & data);120 121bool are_lora_equal(122 const std::vector<common_adapter_lora_info> & l1,123 const std::vector<common_adapter_lora_info> & l2);124 125// get the ids of all enabled loras126std::vector<size_t> lora_get_enabled_ids(const std::vector<common_adapter_lora_info> & loras);127 128//129// server_tokens130//131 132/**133 * server_tokens is a helper to manage the input tokens and image for the server.134 * it is made this way to simplify the logic of KV cache management.135 */136struct server_tokens {137 bool has_mtmd = false;138 139private: // disallow accessing these members directly, risking out-of-sync140 141 // map a **start** index in tokens to the image chunk142 // note: the order need to be in-sync with tokens143 std::map<size_t, mtmd::input_chunk_ptr> map_idx_to_media;144 145 // list of tokens146 // if the token is LLAMA_TOKEN_NULL, it indicates that this position is occupied by media chunk147 // otherwise, it is a normal text token148 // note: a non-text chunk can occupy multiple tokens (aka memory cells) in the token list149 // note(2): for M-RoPE, an image can occupy different number of pos; do not assume 1-to-1 mapping tokens <-> pos150 llama_tokens tokens;151 152 // for ex. with input of 5 text tokens and 2 images (each image occupies 3 tokens and 2 pos):153 // [0] [1] [2] [3] [4] [img0] [img0] [img0] [img1] [img1] [img1]154 // idx 0 1 2 3 4 5 6 7 8 9 10155 // pos 0 1 2 3 4 5 5 5 7 7 7156 // map_idx_to_media will contain: {5, img0}, {8, img1}157 158public:159 server_tokens() = default;160 ~server_tokens() = default;161 162 // Prevent copying163 // TODO: server_tokens should be copyable - remove this:164 server_tokens(const server_tokens&) = delete;165 server_tokens& operator=(const server_tokens&) = delete;166 167 // Allow moving (usually implicitly generated if members are movable)168 server_tokens(server_tokens&&) = default;169 server_tokens& operator=(server_tokens&&) = default;170 171 // Allow accessing elements using [] operator172 llama_token operator[](size_t index) { return tokens[index]; }173 const llama_token& operator[](size_t index) const { return tokens[index]; }174 175 server_tokens(mtmd::input_chunks & mtmd_chunks, bool has_mtmd);176 server_tokens(const llama_tokens & tokens, bool has_mtmd);177 178 // for debugging179 std::string str() const;180 181 // the next position after n_tokens. if n_tokens < 0, return the next position after all tokens.182 llama_pos pos_next(int64_t n_tokens = -1) const;183 184 // number of tokens with position < max_pos185 size_t size_up_to_pos(llama_pos max_pos) const;186 187 const mtmd::input_chunk_ptr & find_chunk(size_t idx) const;188 189 // find next media chunk after idx190 // returns a pair of pointer to the chunk (nullptr if not found) and its start index in tokens191 std::pair<const mtmd::input_chunk_ptr *, size_t> find_next_media_chunk(size_t idx) const;192 193 void push_back(llama_token tok);194 195 // will create a copy of the chunk if it contains non-text data196 void push_back(const mtmd_input_chunk * chunk);197 198 // appends server tokens, updates the media map. copies media chunks.199 void push_back(server_tokens & tokens);200 201 // for compatibility with context shift and prompt truncation202 void insert(const llama_tokens & inp_tokens);203 204 // for compatibility with speculative decoding, ctx shift, slot save/load205 const llama_tokens & get_tokens() const;206 207 llama_tokens get_text_tokens() const;208 209 // for compatibility with speculative decoding210 void set_token(llama_pos pos, llama_token id);211 212 size_t size() const { return tokens.size(); }213 214 bool empty() const { return tokens.empty(); }215 216 // true if the sequence actually contains image/audio chunks.217 bool has_media() const { return !map_idx_to_media.empty(); }218 219 void clear() {220 map_idx_to_media.clear();221 tokens.clear();222 }223 224 void keep_first(size_t n);225 226 std::string detokenize(const llama_context * ctx, bool special) const;227 228 size_t get_common_prefix(const server_tokens & b) const;229 230 // split the tokens into message spans, skipping over media chunks231 common_chat_msg_spans find_message_spans(const common_chat_msg_delimiters & delims) const;232 233 // make sure all text tokens are within the vocab range234 bool validate(const struct llama_context * ctx) const;235 236 server_tokens clone() const;237};238 239 240//241// tokenizer and input processing utils242//243 244bool json_is_array_of_numbers(const json & data);245 246// is array having BOTH numbers & strings?247bool json_is_array_of_mixed_numbers_strings(const json & data);248 249// does array have any individual integers/tokens?250bool json_is_array_and_contains_numbers(const json & data);251 252// get value by path(key1 / key2)253json json_get_nested_values(const std::vector<std::string> & paths, const json & js);254 255/**256 * this handles 2 cases:257 * - only string, example: "string"258 * - mixed string and tokens, example: [12, 34, "string", 56, 78]259 */260llama_tokens tokenize_mixed(const llama_vocab * vocab, const json & json_prompt, bool add_special, bool parse_special);261 262// return the last index of character that can form a valid string263// if the last character is potentially cut in half, return the index before the cut264// if validate_utf8(text) == text.size(), then the whole text is valid utf8265size_t validate_utf8(const std::string& text);266 267// process mtmd prompt, return the server_tokens containing both text tokens and media chunks268// if is_placeholder is true, the media chunk will be treated as placeholder for counting tokens; the output tokens are not usable for actual inference (e.g. for submitting a task to server_queue)269server_tokens process_mtmd_prompt(mtmd_context * mctx, const std::string & prompt, const std::vector<raw_buffer> & files, bool is_placeholder = false);270 271/**272 * break the input "prompt" object into multiple prompt if needed, then tokenize them273 * this supports these cases:274 * - "prompt": "string"275 * - "prompt": [12, 34, 56]276 * - "prompt": [12, 34, "string", 56, 78]277 * - "prompt": { "prompt_string": "string", "multimodal_data": [ "base64" ] }278 * and multiple prompts (multi-tasks):279 * - "prompt": ["string1", "string2"]280 * - "prompt": ["string1", [12, 34, 56]]281 * - "prompt": [[12, 34, 56], [78, 90, 12]]282 * - "prompt": [[12, 34, "string", 56, 78], [12, 34, 56], { "prompt_string": "string", "multimodal_data": [ "base64" ]}]283 */284std::vector<server_tokens> tokenize_input_prompts(285 const llama_vocab * vocab,286 mtmd_context * mctx,287 const json & json_prompt,288 bool add_special,289 bool parse_special);290 291//292// OAI utils293//294 295// global server parameters for chat formatting / parsing296struct server_chat_params {297 bool use_jinja;298 bool prefill_assistant;299 common_reasoning_format reasoning_format;300 std::map<std::string, std::string> chat_template_kwargs; // mapping key --> json value301 common_chat_templates_ptr tmpls;302 bool allow_image;303 bool allow_audio;304 bool allow_video;305 bool enable_thinking = true;306 int reasoning_budget = -1;307 std::string reasoning_budget_message;308 std::string media_path;309 bool force_pure_content = false;310};311 312// used by /completions endpoint313json oaicompat_completion_params_parse(const json & body);314 315// used by /chat/completions endpoint316json oaicompat_chat_params_parse(317 json & body, /* openai api json semantics */318 const server_chat_params & opt,319 std::vector<raw_buffer> & out_files);320 321// TODO: move it to server-task.cpp322json format_embeddings_response_oaicompat(323 const json & request,324 const std::string & model_name,325 const json & embeddings,326 bool use_base64 = false);327 328// TODO: move it to server-task.cpp329json format_response_rerank(330 const json & request,331 const std::string & model_name,332 const json & ranks,333 bool is_tei_format,334 std::vector<std::string> & texts,335 int top_n);336 337//338// other utils339//340 341std::vector<llama_token_data> get_token_probabilities(llama_context * ctx, int idx, size_t n_top);342 343std::string safe_json_to_str(const json & data);344 345std::string tokens_to_str(llama_context * ctx, const llama_tokens & tokens);346std::string tokens_to_str(const llama_vocab * vocab, const llama_tokens & tokens);347 348// format incomplete utf-8 multibyte character for output349std::string tokens_to_output_formatted_string(const llama_context * ctx, const llama_token token);350 351// format server-sent event (SSE), return the formatted string to send352// note: if data is a json array, it will be sent as multiple events, one per item353std::string format_oai_sse(const json & data);354 355std::string format_oai_resp_sse(const json & data);356 357// format Anthropic-style SSE with event types358std::string format_anthropic_sse(const json & data);359 360bool is_valid_utf8(const std::string & str);361 362//363// formatting output responses364// TODO: move these to server-task.cpp365//366 367llama_tokens format_prompt_infill(368 const llama_vocab * vocab,369 const json & input_prefix,370 const json & input_suffix,371 const json & input_extra,372 const int n_batch,373 const int n_predict,374 const int n_ctx,375 const bool spm_infill,376 const llama_tokens & tokens_prompt);377 378// format rerank task: [BOS]query[EOS][SEP]doc[EOS].379server_tokens format_prompt_rerank(380 const struct llama_model * model,381 const struct llama_vocab * vocab,382 mtmd_context * mctx,383 const std::string & query,384 const std::string & doc);385 386// simple implementation of a pipe387// used for streaming data between threads388template<typename T>389struct server_pipe {390 std::mutex mutex;391 std::condition_variable cv;392 std::queue<T> queue;393 std::atomic<bool> writer_closed{false};394 std::atomic<bool> reader_closed{false};395 396 // 0 = unbounded (default)397 // > 0, write() drops the oldest item once the queue is full398 size_t max_size = 0;399 400 void close_write() {401 writer_closed.store(true, std::memory_order_relaxed);402 cv.notify_all();403 }404 405 void close_read() {406 reader_closed.store(true, std::memory_order_relaxed);407 cv.notify_all();408 }409 410 // close_on_stop = true: should_stop means the reader is gone for good, so the writer is told the pipe is broken.411 // close_on_stop = false: should_stop is a per-read deadline and further reads still come, so the pipe stays usable.412 bool read(T & output, const std::function<bool()> & should_stop, bool close_on_stop = true) {413 std::unique_lock<std::mutex> lk(mutex);414 constexpr auto poll_interval = std::chrono::milliseconds(500);415 while (true) {416 if (!queue.empty()) {417 output = std::move(queue.front());418 queue.pop();419 return true;420 }421 if (writer_closed.load()) {422 return false; // clean EOF423 }424 if (should_stop && should_stop()) { // a null should_stop means "never stop"425 if (close_on_stop) {426 close_read(); // signal broken pipe to writer427 }428 return false; // cancelled / deadline reached429 }430 cv.wait_for(lk, poll_interval);431 }432 }433 434 bool write(T && data) {435 std::lock_guard<std::mutex> lk(mutex);436 if (reader_closed.load()) {437 return false; // broken pipe438 }439 if (max_size > 0) {440 while (queue.size() >= max_size) {441 queue.pop(); // drop oldest to stay bounded442 }443 }444 queue.push(std::move(data));445 cv.notify_one();446 return true;447 }448};449 