Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
common.h1163 linesDownload Raw Back to common
1// Various helper functions and utilities2 3#pragma once4 5#include "llama-cpp.h"6 7#include "ggml-opt.h"8#include "ggml.h"9#include "llama.h"10 11#include <set>12#include <sstream>13#include <string>14#include <string_view>15#include <vector>16#include <map>17#include <algorithm>18#include <fstream>19 20#if defined(_WIN32) && !defined(_WIN32_WINNT)21#define _WIN32_WINNT 0x0A0022#endif23 24#ifdef _WIN3225#define DIRECTORY_SEPARATOR '\\'26#else27#define DIRECTORY_SEPARATOR '/'28#endif // _WIN3229 30#define COM_DBG(fmt, ...) LOG_DBG("cmn  %12.*s: " fmt, 12, __func__, __VA_ARGS__)31#define COM_TRC(fmt, ...) LOG_TRC("cmn  %12.*s: " fmt, 12, __func__, __VA_ARGS__)32#define COM_INF(fmt, ...) LOG_INF("cmn  %12.*s: " fmt, 12, __func__, __VA_ARGS__)33#define COM_WRN(fmt, ...) LOG_WRN("cmn  %12.*s: " fmt, 12, __func__, __VA_ARGS__)34#define COM_ERR(fmt, ...) LOG_ERR("cmn  %12.*s: " fmt, 12, __func__, __VA_ARGS__)35#define COM_CNT(fmt, ...) LOG_CNT(""              fmt,               __VA_ARGS__)36 37#define die(msg)          do { fputs("error: " msg "\n", stderr);                exit(1); } while (0)38#define die_fmt(fmt, ...) do { fprintf(stderr, "error: " fmt "\n", __VA_ARGS__); exit(1); } while (0)39 40struct common_time_meas {41    common_time_meas(int64_t & t_acc, bool disable = false);42    ~common_time_meas();43 44    const int64_t t_start_us;45 46    int64_t & t_acc;47};48 49struct common_adapter_lora_info {50    std::string path;51    float scale;52 53    std::string task_name;54    std::string prompt_prefix;55 56    struct llama_adapter_lora * ptr;57};58 59using llama_tokens = std::vector<llama_token>;60 61struct common_control_vector_load_info;62 63//64// CPU utils65//66 67struct common_cpu_params {68    int      n_threads                   = -1;69    bool     cpumask[GGML_MAX_N_THREADS] = {false}; // CPU affinity mask.70    bool     mask_valid                  = false;   // Default: any CPU71    enum ggml_sched_priority  priority   = GGML_SCHED_PRIO_NORMAL;  // Scheduling prio : (0 - normal, 1 - medium, 2 - high, 3 - realtime)72    bool     strict_cpu                  = false;   // Use strict CPU placement73    uint32_t poll                        = 50;      // Polling (busywait) level (0 - no polling, 100 - mostly polling)74};75 76int32_t common_cpu_get_num_physical_cores();77int32_t common_cpu_get_num_math();78 79//80// Common params81//82 83enum llama_example {84    LLAMA_EXAMPLE_BATCHED,85    LLAMA_EXAMPLE_DEBUG,86    LLAMA_EXAMPLE_COMMON,87    LLAMA_EXAMPLE_SPECULATIVE,88    LLAMA_EXAMPLE_COMPLETION,89    LLAMA_EXAMPLE_CLI,90    LLAMA_EXAMPLE_EMBEDDING,91    LLAMA_EXAMPLE_PERPLEXITY,92    LLAMA_EXAMPLE_RETRIEVAL,93    LLAMA_EXAMPLE_PASSKEY,94    LLAMA_EXAMPLE_IMATRIX,95    LLAMA_EXAMPLE_BENCH,96    LLAMA_EXAMPLE_SERVER,97    LLAMA_EXAMPLE_CVECTOR_GENERATOR,98    LLAMA_EXAMPLE_EXPORT_LORA,99    LLAMA_EXAMPLE_MTMD,100    LLAMA_EXAMPLE_LOOKUP,101    LLAMA_EXAMPLE_PARALLEL,102    LLAMA_EXAMPLE_TTS,103    LLAMA_EXAMPLE_DIFFUSION,104    LLAMA_EXAMPLE_FINETUNE,105    LLAMA_EXAMPLE_FIT_PARAMS,106    LLAMA_EXAMPLE_RESULTS,107    LLAMA_EXAMPLE_EXPORT_GRAPH_OPS,108    LLAMA_EXAMPLE_DOWNLOAD,109    LLAMA_EXAMPLE_TOKENIZE,110 111    LLAMA_EXAMPLE_COUNT,112};113 114enum common_sampler_type {115    COMMON_SAMPLER_TYPE_NONE        = 0,116    COMMON_SAMPLER_TYPE_DRY         = 1,117    COMMON_SAMPLER_TYPE_TOP_K       = 2,118    COMMON_SAMPLER_TYPE_TOP_P       = 3,119    COMMON_SAMPLER_TYPE_MIN_P       = 4,120  //COMMON_SAMPLER_TYPE_TFS_Z       = 5,121    COMMON_SAMPLER_TYPE_TYPICAL_P   = 6,122    COMMON_SAMPLER_TYPE_TEMPERATURE = 7,123    COMMON_SAMPLER_TYPE_XTC         = 8,124    COMMON_SAMPLER_TYPE_INFILL      = 9,125    COMMON_SAMPLER_TYPE_PENALTIES   = 10,126    COMMON_SAMPLER_TYPE_TOP_N_SIGMA = 11,127    COMMON_SAMPLER_TYPE_ADAPTIVE_P  = 12,128};129 130// dimensionality reduction methods, used by cvector-generator131enum dimre_method {132    DIMRE_METHOD_PCA,133    DIMRE_METHOD_MEAN,134};135 136enum common_conversation_mode {137    COMMON_CONVERSATION_MODE_DISABLED = 0,138    COMMON_CONVERSATION_MODE_ENABLED  = 1,139    COMMON_CONVERSATION_MODE_AUTO     = 2,140};141 142enum common_grammar_trigger_type {143    COMMON_GRAMMAR_TRIGGER_TYPE_TOKEN,144    COMMON_GRAMMAR_TRIGGER_TYPE_WORD,145    COMMON_GRAMMAR_TRIGGER_TYPE_PATTERN,146    COMMON_GRAMMAR_TRIGGER_TYPE_PATTERN_FULL,147};148 149struct common_grammar_trigger {150    common_grammar_trigger_type type;151    std::string value;152    llama_token token = LLAMA_TOKEN_NULL;153};154 155enum common_params_sampling_config : uint64_t {156    COMMON_PARAMS_SAMPLING_CONFIG_SAMPLERS        = 1 << 0,157    COMMON_PARAMS_SAMPLING_CONFIG_TOP_K           = 1 << 1,158    COMMON_PARAMS_SAMPLING_CONFIG_TOP_P           = 1 << 2,159    COMMON_PARAMS_SAMPLING_CONFIG_MIN_P           = 1 << 3,160    COMMON_PARAMS_SAMPLING_CONFIG_XTC_PROBABILITY = 1 << 4,161    COMMON_PARAMS_SAMPLING_CONFIG_XTC_THRESHOLD   = 1 << 5,162    COMMON_PARAMS_SAMPLING_CONFIG_TEMP            = 1 << 6,163    COMMON_PARAMS_SAMPLING_CONFIG_PENALTY_LAST_N  = 1 << 7,164    COMMON_PARAMS_SAMPLING_CONFIG_PENALTY_REPEAT  = 1 << 8,165    COMMON_PARAMS_SAMPLING_CONFIG_MIROSTAT        = 1 << 9,166    COMMON_PARAMS_SAMPLING_CONFIG_MIROSTAT_TAU    = 1 << 10,167    COMMON_PARAMS_SAMPLING_CONFIG_MIROSTAT_ETA    = 1 << 11,168};169 170enum common_speculative_type {171    COMMON_SPECULATIVE_TYPE_NONE,          // no speculative decoding172    COMMON_SPECULATIVE_TYPE_DRAFT_SIMPLE,  // standalone draft model speculative decoding173    COMMON_SPECULATIVE_TYPE_DRAFT_EAGLE3,  // Eagle3 speculative decoding174    COMMON_SPECULATIVE_TYPE_DRAFT_MTP,     // Multi-token prediction175    COMMON_SPECULATIVE_TYPE_DRAFT_DFLASH,  // DFlash speculative decoding176    COMMON_SPECULATIVE_TYPE_DRAFT_DSPARK,  // DSpark speculative decoding (DFlash + Markov head)177    COMMON_SPECULATIVE_TYPE_NGRAM_SIMPLE,  // simple self-speculative decoding based on n-grams178    COMMON_SPECULATIVE_TYPE_NGRAM_MAP_K,   // self-speculative decoding with n-gram keys only179    COMMON_SPECULATIVE_TYPE_NGRAM_MAP_K4V, // self-speculative decoding with n-gram keys and 4 m-gram values180    COMMON_SPECULATIVE_TYPE_NGRAM_MOD,181    COMMON_SPECULATIVE_TYPE_NGRAM_CACHE,   // self-speculative decoding with 3-level n-gram cache182    COMMON_SPECULATIVE_TYPE_COUNT          // number of types, unknown type183};184 185// Grammar type enumeration186enum common_grammar_type {187    COMMON_GRAMMAR_TYPE_NONE,           // no grammar set188    COMMON_GRAMMAR_TYPE_USER,           // user-provided GBNF (--grammar / "grammar" API field)189    COMMON_GRAMMAR_TYPE_OUTPUT_FORMAT,  // auto-generated from JSON schema (--json-schema / "json_schema" API field)190    COMMON_GRAMMAR_TYPE_TOOL_CALLS,     // auto-generated by chat template parser for function calling191};192 193// Grammar variant struct with type and grammar string194struct common_grammar {195    common_grammar_type type = COMMON_GRAMMAR_TYPE_NONE;196    std::string grammar;197 198    // Default constructor - no grammar199    common_grammar() = default;200 201    // Constructor with type and grammar string202    common_grammar(common_grammar_type t, std::string g) : type(t), grammar(std::move(g)) {203        GGML_ASSERT(type != COMMON_GRAMMAR_TYPE_NONE || !grammar.empty());204    }205 206    // Check if a grammar is set207    bool empty() const { return type == COMMON_GRAMMAR_TYPE_NONE || grammar.empty(); }208};209 210// Returns the raw grammar string, or empty string if no grammar is set.211inline const std::string & common_grammar_value(const common_grammar & g) {212    return g.grammar;213}214 215// Returns true when the generation_prompt should be prefilled into the grammar sampler.216// Only output-format and tool-call grammars need prefill; user-supplied grammars must not be prefilled.217inline bool common_grammar_needs_prefill(const common_grammar & g) {218    return g.type == COMMON_GRAMMAR_TYPE_OUTPUT_FORMAT219        || g.type == COMMON_GRAMMAR_TYPE_TOOL_CALLS;220}221 222// sampling parameters223struct common_params_sampling {224    uint32_t seed = LLAMA_DEFAULT_SEED; // the seed used to initialize llama_sampler225 226    int32_t n_prev             = 64;     // number of previous tokens to remember227    int32_t n_probs            = 0;      // if greater than 0, output the probabilities of top n_probs tokens.228    int32_t min_keep           = 0;      // 0 = disabled, otherwise samplers should return at least min_keep tokens229    int32_t top_k              = 40;     // <= 0 to use vocab size230    float   top_p              = 0.95f;  // 1.0 = disabled231    float   min_p              = 0.05f;  // 0.0 = disabled232    float   xtc_probability    = 0.00f;  // 0.0 = disabled233    float   xtc_threshold      = 0.10f;  // > 0.5 disables XTC234    float   typ_p              = 1.00f;  // typical_p, 1.0 = disabled235    float   temp               = 0.80f;  // <= 0.0 to sample greedily, 0.0 to not output probabilities236    float   dynatemp_range     = 0.00f;  // 0.0 = disabled237    float   dynatemp_exponent  = 1.00f;  // controls how entropy maps to temperature in dynamic temperature sampler238    int32_t penalty_last_n     = 64;     // last n tokens to penalize (0 = disable penalty)239    float   penalty_repeat     = 1.00f;  // 1.0 = disabled240    float   penalty_freq       = 0.00f;  // 0.0 = disabled241    float   penalty_present    = 0.00f;  // 0.0 = disabled242    float   dry_multiplier     = 0.0f;   // 0.0 = disabled;      DRY repetition penalty for tokens extending repetition:243    float   dry_base           = 1.75f;  // 0.0 = disabled;      multiplier * base ^ (length of sequence before token - allowed length)244    int32_t dry_allowed_length = 2;      // tokens extending repetitions beyond this receive penalty245    int32_t dry_penalty_last_n = 64;     // how many tokens to scan for repetitions (0 = disable penalty)246    float   adaptive_target    = -1.0f;  // select tokens near this probability (valid range 0.0 to 1.0; negative = disabled)247    float   adaptive_decay     = 0.90f;  // EMA decay for adaptation; history ≈ 1/(1-decay) tokens (0.0 - 0.99)248    int32_t mirostat           = 0;      // 0 = disabled, 1 = mirostat, 2 = mirostat 2.0249    float   top_n_sigma        = -1.00f; // -1.0 = disabled250    float   mirostat_tau       = 5.00f;  // target entropy251    float   mirostat_eta       = 0.10f;  // learning rate252    bool    ignore_eos         = false;253    bool    no_perf            = false;  // disable performance metrics254    bool    timing_per_token   = false;255 256    uint64_t user_sampling_config = 0; // bitfield to track user-specified samplers257 258    std::vector<std::string> dry_sequence_breakers = {"\n", ":", "\"", "*"};     // default sequence breakers for DRY259 260    std::vector<enum common_sampler_type> samplers = {261        COMMON_SAMPLER_TYPE_PENALTIES,262        COMMON_SAMPLER_TYPE_DRY,263        COMMON_SAMPLER_TYPE_TOP_N_SIGMA,264        COMMON_SAMPLER_TYPE_TOP_K,265        COMMON_SAMPLER_TYPE_TYPICAL_P,266        COMMON_SAMPLER_TYPE_TOP_P,267        COMMON_SAMPLER_TYPE_MIN_P,268        COMMON_SAMPLER_TYPE_XTC,269        COMMON_SAMPLER_TYPE_TEMPERATURE,270    };271 272    common_grammar              grammar;      // optional grammar constraint (user / output-format / tool-calls)273    bool                                grammar_lazy = false;274    std::vector<common_grammar_trigger> grammar_triggers; // optional triggers (for lazy grammars)275    std::set<llama_token>               preserved_tokens;276 277    std::vector<llama_logit_bias> logit_bias;     // logit biases to apply278    std::vector<llama_logit_bias> logit_bias_eog; // pre-calculated logit biases for EOG tokens279 280    // The assistant generation prompt already prefilled into the prompt.281    // Fed to the grammar sampler (to advance past pre-existing tokens) and used282    // to determine the reasoning budget sampler's initial state.283    // Only applied when the grammar is of output-format or tool-calls type.284    std::string generation_prompt;285 286    // reasoning budget sampler parameters287    // these are populated by the server/CLI based on chat template params288    int32_t                   reasoning_budget_tokens   = -1;  // -1 = disabled, >= 0 = token budget289    std::vector<llama_token>  reasoning_budget_start;          // start tag token sequence290    std::vector<llama_tokens> reasoning_budget_end;            // end tag token sequences; the first tag is used as the forcing sequence291    std::vector<llama_token>  reasoning_budget_forced;         // forced sequence (message + first end tag)292    std::string               reasoning_budget_message;        // message injected before end tag when budget exhausted293    bool                      reasoning_control = false;       // create the budget sampler on demand so reasoning can be ended at runtime294 295    bool backend_sampling = false;296 297    // print the parameters into a string298    std::string print() const;299};300 301struct common_params_model {302    std::string path        = ""; // model local path303    std::string url         = ""; // model url to download304    std::string hf_repo     = ""; // HF repo305    std::string hf_file     = ""; // HF file306    std::string docker_repo = ""; // Docker repo307 308    std::string get_name() const {309        if (!hf_repo.empty()) {310            return hf_repo;311        }312        if (!docker_repo.empty()) {313            return docker_repo;314        }315        return path;316    }317 318    bool empty() const {319        return get_name().empty();320    }321};322 323// draft-model-based speculative decoding parameters324struct common_params_speculative_draft {325    int32_t n_max = 8; // maximum number of tokens to draft during speculative decoding (optimized for code)326    int32_t n_min = 0; // minimum number of draft tokens to use for speculative decoding327 328    float p_split = 0.1f; // speculative decoding split probability329    float p_min   = 0.75f; // minimum speculative decoding probability (greedy)330 331    bool backend_sampling = true; // offload draft sampling to the backend (default: on)332 333    common_params_model mparams;334 335    llama_context * ctx_tgt = nullptr;336    llama_context * ctx_dft = nullptr;337 338    int32_t n_gpu_layers = -1; // number of layers to store in VRAM for the draft model (-1 - use default)339 340    ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K341    ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V342 343    common_cpu_params cpuparams;344    common_cpu_params cpuparams_batch;345 346    std::vector<ggml_backend_dev_t> devices; // devices to use for offloading347 348    std::vector<llama_model_tensor_buft_override> tensor_buft_overrides;349};350 351struct common_params_speculative_ngram_mod {352    int32_t n_match = 24;353 354    int32_t n_max = 64;355    int32_t n_min = 48;356};357 358struct common_params_speculative_ngram_map {359    uint16_t size_n   = 12; // ngram size for lookup360    uint16_t size_m   = 48; // mgram size for speculative tokens361    uint16_t min_hits = 1;  // minimum hits at ngram/mgram lookup for mgram to be proposed362};363 364struct common_params_speculative_ngram_cache {365    std::string lookup_cache_static;  // path of static ngram cache file for lookup decoding366    std::string lookup_cache_dynamic; // path of dynamic ngram cache file for lookup decoding367};368 369struct common_params_speculative {370    std::vector<enum common_speculative_type> types = { COMMON_SPECULATIVE_TYPE_NONE };371 372    // used by Simple, MTP, Eagle3, etc. - all methods that require some kind of draft model373    common_params_speculative_draft draft;374 375    common_params_speculative_ngram_mod ngram_mod;376    common_params_speculative_ngram_map ngram_simple;377    common_params_speculative_ngram_map ngram_map_k;378    common_params_speculative_ngram_map ngram_map_k4v;379 380    common_params_speculative_ngram_cache ngram_cache;381 382    bool has_dft() const {383        return !draft.mparams.empty();384    }385 386    uint32_t need_n_rs_seq() const {387        bool needs_rs_seq = std::any_of(types.begin(), types.end(), [&](auto t) {388            return t == COMMON_SPECULATIVE_TYPE_DRAFT_MTP || t == COMMON_SPECULATIVE_TYPE_DRAFT_EAGLE3 || t == COMMON_SPECULATIVE_TYPE_DRAFT_DFLASH || t == COMMON_SPECULATIVE_TYPE_DRAFT_DSPARK;389        });390 391        return needs_rs_seq ? draft.n_max : 0u;392    }393};394 395struct common_params_diffusion {396    int32_t steps         = 128;397    bool    visual_mode   = false;398 399    float   eps           = 0;        // epsilon for timesteps400    int32_t block_length  = 0;        // block length for generation401 402    int32_t algorithm     = 4;        // default algorithm: low-confidence403    float   alg_temp      = 0.0f;     // algorithm temperature404 405    float   cfg_scale     = 0;        // classifier-free guidance scale406    bool    add_gumbel_noise = false; // add gumbel noise to the logits if temp > 0.0407};408 409// reasoning API response format (not to be confused as chat template's reasoning format)410// only used by server411enum common_reasoning_format {412    COMMON_REASONING_FORMAT_NONE,413    COMMON_REASONING_FORMAT_AUTO,            // Same as deepseek, using `message.reasoning_content`414    COMMON_REASONING_FORMAT_DEEPSEEK_LEGACY, // Extract thinking tag contents and return as `message.reasoning_content`, or leave inline in <think> tags in stream mode415    COMMON_REASONING_FORMAT_DEEPSEEK,        // Extract thinking tag contents and return as `message.reasoning_content`, including in streaming deltas.416    // do not extend this enum unless you absolutely have to417    // in most cases, use COMMON_REASONING_FORMAT_AUTO418    // see: https://github.com/ggml-org/llama.cpp/pull/15408419};420 421 422struct lr_opt {423    float    lr0          = 1e-5; // learning rate at first epoch424    float    lr_min       = -1;425    float    decay_epochs = -1;   // if >0, the learning rate starts at lr0 and decays to lr_min after this many epochs426    float    scale_epoch  = 0;427    float    wd           = 0;428    unsigned epochs       = 2;429 430    unsigned epoch; // set by optimizer outer (epochs) loop431    // learning rate decay - constant LR per epoch only for now432    float get_lr(float e) const;433    float get_lr() const { return get_lr(epoch); }434    // must call after arg parse, before get_lr435    void init();436};437 438struct ggml_opt_optimizer_params common_opt_lr_pars(void * userdata);439 440struct common_params {441    int32_t n_predict             =    -1; // max. number of new tokens to predict, -1 == no limit442    int32_t n_ctx                 =     0; // context size, 0 == context the model was trained with443    int32_t n_batch               =  2048; // logical batch size for prompt processing (must be >=32 to use BLAS)444    int32_t n_ubatch              =   512; // physical batch size for prompt processing (must be >=32 to use BLAS)445    int32_t n_keep                =     0; // number of tokens to keep from initial prompt446    int32_t n_chunks              =    -1; // max number of chunks to process (-1 = unlimited)447    int32_t n_parallel            =     1; // number of parallel sequences to decode448    int32_t n_sequences           =     1; // number of sequences to decode449    int32_t n_outputs_max         =     0; // max outputs in a batch (0 = n_batch)450    int32_t n_outputs_max_per_seq =     1; // max outputs per sequence451    int32_t grp_attn_n            =     1; // group-attention factor452    int32_t grp_attn_w            =   512; // group-attention width453    int32_t n_print               =    -1; // print token count every n tokens (-1 = disabled)454    float   rope_freq_base        =  0.0f; // RoPE base frequency455    float   rope_freq_scale       =  0.0f; // RoPE frequency scaling factor456    float   yarn_ext_factor       = -1.0f; // YaRN extrapolation mix factor457    float   yarn_attn_factor      = -1.0f; // YaRN magnitude scaling factor458    float   yarn_beta_fast        = -1.0f; // YaRN low correction dim459    float   yarn_beta_slow        = -1.0f; // YaRN high correction dim460    int32_t yarn_orig_ctx         =     0; // YaRN original context length461 462    // offload params463    std::vector<ggml_backend_dev_t> devices; // devices to use for offloading464 465    int32_t n_gpu_layers       = -1;    // number of layers to store in VRAM, -1 is auto, <= -2 is all466    int32_t main_gpu           = 0;     // the GPU that is used for scratch and small tensors467    float   tensor_split[128]  = {0};   // how split tensors should be distributed across GPUs468    bool    fit_params         = true;  // whether to fit unset model/context parameters to free device memory469    bool    fit_params_print   = false; // print the estimated required memory to run the model470    int32_t fit_params_min_ctx = 4096;  // minimum context size to set when trying to reduce memory use471 472    // margin per device in bytes for fitting parameters to free memory:473    std::vector<size_t> fit_params_target = std::vector<size_t>(llama_max_devices(), 1024 * 1024*1024);474 475    enum llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER; // how to split the model across GPUs476    enum llama_load_mode  load_mode  = LLAMA_LOAD_MODE_MMAP; // how to load the model477 478    common_cpu_params cpuparams;479    common_cpu_params cpuparams_batch;480 481    ggml_backend_sched_eval_callback cb_eval = nullptr;482    void * cb_eval_user_data                 = nullptr;483 484    ggml_numa_strategy numa = GGML_NUMA_STRATEGY_DISABLED;485 486    enum llama_rope_scaling_type rope_scaling_type = LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED;487    enum llama_pooling_type      pooling_type      = LLAMA_POOLING_TYPE_UNSPECIFIED; // pooling type for embeddings488    enum llama_attention_type    attention_type    = LLAMA_ATTENTION_TYPE_UNSPECIFIED; // attention type for embeddings489    enum llama_flash_attn_type   flash_attn_type   = LLAMA_FLASH_ATTN_TYPE_AUTO; // whether to use Flash Attention490 491    struct common_params_sampling    sampling;492    struct common_params_speculative speculative;493    struct common_params_diffusion   diffusion;494 495    struct common_params_model model;496 497    std::set<std::string> model_alias;     // model aliases                                                 // NOLINT498    std::set<std::string> model_tags;      // model tags (informational, not used for routing)              // NOLINT499    std::string hf_token             = ""; // HF token (aka bearer token)                                   // NOLINT500    std::string prompt               = "";                                                                  // NOLINT501    std::string system_prompt        = "";                                                                  // NOLINT502    std::string prompt_file          = ""; // store the external prompt file name                           // NOLINT503    std::string path_prompt_cache    = ""; // path to file for saving/loading prompt eval state             // NOLINT504    std::string input_prefix         = ""; // string to prefix user inputs with                             // NOLINT505    std::string input_suffix         = ""; // string to suffix user inputs with                             // NOLINT506    std::string logits_file          = ""; // file for saving *all* logits                                  // NOLINT507    std::string path_prompts_log_dir = ""; // directory with logged prompts                                 // NOLINT508 509    // llama-debug specific options510    std::string logits_output_dir = "data"; // directory for saving logits output files                     // NOLINT511    bool        save_logits       = false;  // whether to save logits to files                              // NOLINT512    std::vector<std::string> tensor_filter; // filter tensor names for debug output (regex)                 // NOLINT513 514    std::vector<std::string> in_files;   // all input files515    std::vector<std::string> antiprompt; // strings upon which more user input is prompted (a.k.a. reverse prompts)516    std::vector<llama_model_kv_override> kv_overrides;517    std::vector<llama_model_tensor_buft_override> tensor_buft_overrides;518 519    bool lora_init_without_apply = false; // only load lora to memory, but do not apply it to ctx (user can manually apply lora later using llama_adapter_lora_apply)520    std::vector<common_adapter_lora_info> lora_adapters; // lora adapter path with user defined scale521 522    std::vector<common_control_vector_load_info> control_vectors; // control vector with user defined scale523 524    int32_t verbosity                  = 3;  // LOG_LEVEL_INFO525    int32_t control_vector_layer_start = -1; // layer range for control vector526    int32_t control_vector_layer_end   = -1; // layer range for control vector527    bool    offline                    = false;528 529    int32_t ppl_stride      = 0;     // stride for perplexity calculations. If left at 0, the pre-existing approach will be used.530    int32_t ppl_output_type = 0;     // = 0 -> ppl output is as usual, = 1 -> ppl output is num_tokens, ppl, one per line531                                     //                                       (which is more convenient to use for plotting)532                                     //533    bool   hellaswag        = false; // compute HellaSwag score over random tasks from datafile supplied in prompt534    size_t hellaswag_tasks  = 400;   // number of tasks to use when computing the HellaSwag score535 536    bool   winogrande       = false; // compute Winogrande score over random tasks from datafile supplied in prompt537    size_t winogrande_tasks = 0;     // number of tasks to use when computing the Winogrande score. If 0, all tasks will be computed538 539    bool   multiple_choice  = false;  // compute TruthfulQA score over random tasks from datafile supplied in prompt540    size_t multiple_choice_tasks = 0; // number of tasks to use when computing the TruthfulQA score. If 0, all tasks will be computed541 542    bool   kl_divergence    = false; // compute KL divergence543 544    bool check             = false; // check rather than generate results for llama-results545 546    bool usage             = false; // print usage547    bool completion        = false; // print source-able completion script548    bool use_color         = false; // use color to distinguish generations and inputs549    bool special           = false; // enable special token output550    bool interactive       = false; // interactive mode551    bool interactive_first = false; // wait for user input immediately552    bool prompt_cache_all  = false; // save user input and generations to prompt cache553    bool prompt_cache_ro   = false; // open the prompt cache read-only and do not update it554 555    bool escape            = true;  // escape "\n", "\r", "\t", "\'", "\"", and "\\"556    bool multiline_input   = false; // reverse the usage of `\`557    bool simple_io         = false; // improves compatibility with subprocesses and limited consoles558    bool cont_batching     = true;  // insert new sequences for decoding on-the-fly559    bool no_perf           = false; // disable performance metrics560    bool show_timings      = true;  // show timing information on CLI561    bool ctx_shift         = false; // context shift on infinite text generation562    bool swa_full          = false; // use full-size SWA cache (https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)563    bool kv_unified        = false; // enable unified KV cache564 565    bool input_prefix_bos  = false; // prefix BOS to user inputs, preceding input_prefix566    bool verbose_prompt    = false; // print prompt tokens before generation567    bool display_prompt    = true;  // print prompt before generation568    bool no_kv_offload     = false; // disable KV offloading569    bool warmup            = true;  // warmup run570    bool check_tensors     = false; // validate tensor data571    bool no_op_offload     = false; // globally disable offload host tensor operations to device572    bool no_extra_bufts    = false; // disable extra buffer types (used for weight repacking)573    bool no_host           = false; // bypass host buffer allowing extra buffers to be used574 575    bool single_turn       = false; // single turn chat conversation576 577    ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K578    ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V579 580    common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;581 582    // multimodal models (see tools/mtmd)583    struct common_params_model mmproj;584    bool mmproj_use_gpu = true;     // use GPU for multimodal model585    bool no_mmproj = false;         // explicitly disable multimodal model586    std::vector<std::string> image; // path to image file(s) ; TODO: change the name to "media"587    int image_min_tokens = -1;588    int image_max_tokens = -1;589    int mtmd_batch_max_tokens = 1024;590 591    // finetune592    struct lr_opt lr;593    enum ggml_opt_optimizer_type optimizer = GGML_OPT_OPTIMIZER_TYPE_ADAMW;594    float val_split = 0.05f; // fraction of the data used for the validation set595 596    // embedding597    bool embedding         = false; // get only sentence embedding598    int32_t embd_normalize = 2;     // normalisation for embeddings (-1=none, 0=max absolute int16, 1=taxicab, 2=euclidean, >2=p-norm)599    std::string embd_out   = "";    // empty = default, "array" = [[],[]...], "json" = openai style, "json+" = same "json" + cosine similarity matrix600    std::string embd_sep   = "\n";  // separator of embeddings601    std::string cls_sep    = "\t";  // separator of classification sequences602 603    // server params604    int32_t port                = 8080;          // server listens on this network port605    bool    reuse_port          = false;         // allow multiple sockets to bind to the same port606    int32_t timeout_read        = 3600;          // http read timeout in seconds607    int32_t timeout_write       = timeout_read;  // http write timeout in seconds608    int32_t sse_ping_interval   = 30;            // SSE ping interval in seconds609    int32_t n_threads_http      = -1;    // number of threads to process HTTP requests (TODO: support threadpool)610    int32_t n_cache_reuse       = 0;     // min chunk size to reuse from the cache via KV shifting611    bool    cache_prompt        = true;  // whether to enable prompt caching612    bool    cache_idle_slots    = true;  // save and clear idle slots upon starting a new task613    int32_t n_ctx_checkpoints   = 32;    // max number of context checkpoints per slot614    int32_t checkpoint_min_step = 8192;  // minimum spacing between context checkpoints615    int32_t cache_ram_mib       = 8192;  // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.616 617    std::string hostname      = "127.0.0.1";618    std::string public_path   = "";                                                                         // NOLINT619    std::string api_prefix    = "";                                                                         // NOLINT620    std::string chat_template = "";                                                                         // NOLINT621    bool use_jinja = true;                                                                                  // NOLINT622 623    // server CORS params624    std::string cors_origins = "*";625    std::string cors_methods = "GET, POST, DELETE, OPTIONS";626    std::string cors_headers = "*";627    bool cors_credentials = true;628    bool cors_origins_explicit = false; // for --agent option629 630    bool enable_chat_template = true;631    bool force_pure_content_parser = false;632    common_reasoning_format reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;633    int enable_reasoning = -1; // -1 = auto, 0 = disable, 1 = enable634    bool prefill_assistant = true; // if true, any trailing assistant message will be prefilled into the response635    int sleep_idle_seconds = -1;   // if >0, server will sleep after this many seconds of idle time636 637    std::vector<std::string> api_keys;638 639    std::string ssl_file_key  = "";                                                                         // NOLINT640    std::string ssl_file_cert = "";                                                                         // NOLINT641 642    std::map<std::string, std::string> default_template_kwargs;643 644    // CLI params645    std::string server_base; // if set, connect to this server instead of starting a new one646 647    // UI configs648    bool ui = true;649    bool ui_mcp_proxy = true;650    std::string ui_config_json;651 652    // "advanced" endpoints are disabled by default for better security653    bool endpoint_slots   = true;654    bool endpoint_props   = false; // only control POST requests, not GET655    bool endpoint_metrics = false;656 657    // enable built-in tools (enabled by default)658    std::vector<std::string> server_tools = {"all"};659    std::string server_tools_runtime;660 661    // MCP server configs (Cursor-compatible JSON)662    std::string mcp_servers_config;   // path to JSON file with MCP server definitions663    std::string mcp_servers_json;     // inline JSON with MCP server definitions664 665    // router server configs666    std::string models_dir    = "";     // directory containing models for the router server667    std::string models_preset = "";     // directory containing model presets for the router server668    int models_max = 4;                 // maximum number of models to load simultaneously669    bool models_autoload = true;        // automatically load models when requested via the router server670    std::string models_preset_hf = "";  // show a warning about remote presets on router loaded (if not empty)671 672    bool log_json = false;673 674    std::string slot_save_path;675    std::string media_path; // path to directory for loading media files676 677    float slot_prompt_similarity = 0.1f;678 679    // batched-bench params680    bool is_pp_shared   = false;681    bool is_tg_separate = false;682 683    std::vector<int32_t> n_pp;684    std::vector<int32_t> n_tg;685    std::vector<int32_t> n_pl;686 687    // retrieval params688    std::vector<std::string> context_files; // context files to embed689 690    int32_t chunk_size = 64; // chunk size for context embedding691 692    std::string chunk_separator = "\n"; // chunk separator for context embedding693 694    // passkey params695    int32_t n_junk = 250; // number of times to repeat the junk text696    int32_t i_pos  = -1;  // position of the passkey in the junk text697 698    // imatrix params699    int32_t n_out_freq  = 10; // output the imatrix every n_out_freq iterations700    int32_t n_save_freq =  0; // save the imatrix every n_save_freq iterations701    int32_t i_chunk     =  0; // start processing from this chunk702    int8_t  imat_dat    =  0; // whether the legacy imatrix.dat format should be output (gguf <= 0 < dat)703 704    bool process_output  = false; // collect data for the output tensor705    bool compute_ppl     = true;  // whether to compute perplexity706    bool show_statistics = false; // show imatrix statistics per tensor707    bool parse_special   = false; // whether to parse special tokens during imatrix tokenization708 709    // cvector-generator params710    int n_pca_batch = 100;711    int n_pca_iterations = 1000;712    dimre_method cvector_dimre_method = DIMRE_METHOD_PCA;713    std::string cvector_positive_file = "tools/cvector-generator/positive.txt";714    std::string cvector_negative_file = "tools/cvector-generator/negative.txt";715 716    bool spm_infill = false; // suffix/prefix/middle pattern for infill717 718    // batched-bench params719    bool batched_bench_output_jsonl = false;720 721    // tokenize params722    bool tokenize_ids        = false; // if true, only print the token IDs723    bool tokenize_stdin      = false; // if true, read the prompt from stdin724    bool tokenize_no_bos     = false; // if true, do not add the BOS token725    bool tokenize_show_count = false; // if true, print the total token count726 727    // common params728    std::string out_file; // output filename for all example programs729    // optional callback for model loading progress and cancellation:730    // called with a progress value between 0.0 and 1.0.731    // return false from callback to abort model loading or true to continue732    llama_progress_callback load_progress_callback = NULL;733    void *                  load_progress_callback_user_data = NULL;734    bool no_alloc = false; // Don't allocate model buffers735 736    // TTS params737    std::string tts_lang = "";738    std::string tts_speaker_file = "";739 740    bool is_gen_docs = false; // whether we are running inside llama-gen-docs741};742 743// call once at the start of a program if it uses libcommon744// initializes the logging system and prints info about the build745void common_init();746 747void common_params_print_info(const common_params & params, bool print_devices = true);748std::string common_params_get_system_info(const common_params & params);749 750bool parse_cpu_range(const std::string & range, bool(&boolmask)[GGML_MAX_N_THREADS]);751bool parse_cpu_mask(const std::string & mask, bool(&boolmask)[GGML_MAX_N_THREADS]);752void postprocess_cpu_params(common_cpu_params & cpuparams, const common_cpu_params * role_model = nullptr);753bool set_process_priority(enum ggml_sched_priority prio);754 755//756// String utils757//758 759#ifdef __GNUC__760#    if defined(__MINGW32__) && !defined(__clang__)761#        define LLAMA_COMMON_ATTRIBUTE_FORMAT(...) __attribute__((format(gnu_printf, __VA_ARGS__)))762#    else763#        define LLAMA_COMMON_ATTRIBUTE_FORMAT(...) __attribute__((format(printf, __VA_ARGS__)))764#    endif765#else766#    define LLAMA_COMMON_ATTRIBUTE_FORMAT(...)767#endif768 769LLAMA_COMMON_ATTRIBUTE_FORMAT(1, 2)770std::string string_format(const char * fmt, ...);771 772std::string string_strip(const std::string & str);773std::string string_get_sortable_timestamp();774std::string string_lcs(std::string_view a, std::string_view b);775 776std::string string_join(const std::vector<std::string> & values, const std::string & separator);777std::vector<std::string> string_split(const std::string & str, const std::string & delimiter);778std::string string_repeat(const std::string & str, size_t n);779 780void string_replace_all(std::string & s, const std::string & search, const std::string & replace);781 782std::string regex_escape(const std::string & s);783 784template<class T>785static std::vector<T> string_split(const std::string & str, char delim) {786    static_assert(!std::is_same<T, std::string>::value, "Please use the specialized version for std::string");787    std::vector<T> values;788    std::istringstream str_stream(str);789    std::string token;790    while (std::getline(str_stream, token, delim)) {791        T value;792        std::istringstream token_stream(token);793        token_stream >> value;794        values.push_back(value);795    }796    return values;797}798 799template<>800inline std::vector<std::string> string_split<std::string>(const std::string & str, char delim)801{802    std::vector<std::string> parts;803    size_t begin_pos = 0;804    size_t delim_pos = str.find(delim);805    while (delim_pos != std::string::npos) {806        std::string part = str.substr(begin_pos, delim_pos - begin_pos);807        parts.emplace_back(part);808        begin_pos = delim_pos + 1;809        delim_pos = str.find(delim, begin_pos);810    }811    parts.emplace_back(str.substr(begin_pos));812    return parts;813}814 815// remove when moving to c++20816inline bool string_starts_with(std::string_view str, std::string_view prefix) {817    return str.size() >= prefix.size() &&818           str.compare(0, prefix.size(), prefix) == 0;819}820 821// remove when moving to c++20822inline bool string_starts_with(std::string_view str, char prefix) {823    return !str.empty() && str.front() == prefix;824}825 826// remove when moving to c++20827inline bool string_ends_with(std::string_view str, std::string_view suffix) {828    return str.size() >= suffix.size() &&829           str.compare(str.size() - suffix.size(), suffix.size(), suffix) == 0;830}831 832inline bool string_remove_suffix(std::string & str, std::string_view suffix) {833    if (string_ends_with(str, suffix)) {834        str.resize(str.size() - suffix.size());835        return true;836    }837    return false;838}839 840inline size_t string_find_partial_stop(std::string_view str, std::string_view stop) {841    if (!str.empty() && !stop.empty()) {842        const size_t max_len = std::min(str.size(), stop.size());843        const char last_char = str.back();844        for (size_t len = max_len; len > 0; --len) {845            if (stop[len - 1] == last_char) {846                if (string_ends_with(str, stop.substr(0, len))) {847                    return str.size() - len;848                }849            }850        }851    }852    return std::string::npos;853}854 855bool string_parse_kv_override(const char * data, std::vector<llama_model_kv_override> & overrides);856void string_process_escapes(std::string & input);857 858std::string string_from(bool value);859std::string string_from(const std::vector<int> & values);860std::string string_from(const struct llama_context * ctx, const std::vector<llama_token> & tokens);861std::string string_from(const struct llama_context * ctx, const struct llama_batch & batch);862 863bool glob_match(const std::string & pattern, const std::string & str);864 865//866// Environment utils867//868 869// portable environment access, an unset variable reads as an empty string870// and setting an empty value unsets the variable871std::string common_get_env(const std::string & name);872void        common_set_env(const std::string & name, const std::string & value);873 874//875// Filesystem utils876//877 878bool fs_validate_filename(const std::string & filename, bool allow_subdirs = false);879bool fs_create_directory_with_parents(const std::string & path);880bool fs_is_directory(const std::string & path);881 882std::string fs_get_cache_directory();883std::string fs_get_cache_file(const std::string & filename);884 885struct common_file_info {886    std::string path;887    std::string name;888    size_t      size = 0; // in bytes889    bool        is_dir = false;890};891std::vector<common_file_info> fs_list(const std::string & path, bool include_directories);892 893// fs open, also handle UTF8 on Windows894std::ifstream fs_open_ifstream(const std::string & fname, std::ios_base::openmode mode);895 896//897// TTY utils898//899 900// Auto-detect if colors can be enabled based on terminal and environment901bool tty_can_use_colors();902 903//904// Model utils905//906 907struct common_sampler;908 909// note: defines the model, context, samplers, ets. lifetimes910struct common_init_result {911    common_init_result(common_params & params, bool model_only = false);912    ~common_init_result();913 914    llama_model * model();915    llama_context * context();916 917    common_sampler * sampler(llama_seq_id seq_id);918    void reset_samplers();919 920    std::vector<llama_adapter_lora_ptr> & lora();921 922private:923    struct impl;924    std::unique_ptr<impl> pimpl;925};926 927using common_init_result_ptr = std::unique_ptr<common_init_result>;928 929common_init_result_ptr common_init_from_params(common_params & params, bool model_only = false);930 931struct llama_model_params     common_model_params_to_llama  (      common_params & params);932struct llama_context_params   common_context_params_to_llama(const common_params & params);933struct ggml_threadpool_params ggml_threadpool_params_from_cpu_params(const common_cpu_params & params);934 935// clear LoRA adapters from context, then apply new list of adapters936void common_set_adapter_lora(struct llama_context * ctx, std::vector<common_adapter_lora_info> & lora);937 938// model endpoint from env939std::string common_get_model_endpoint();940 941// for testing purposes942char * common_get_model_or_exit(int, char*[]);943 944//945// Context utils946//947 948enum common_context_seq_rm_type {949    COMMON_CONTEXT_SEQ_RM_TYPE_NO           = 0, // seq_rm not supported (e.g. no memory module)950    COMMON_CONTEXT_SEQ_RM_TYPE_PART         = 1, // can seq_rm partial sequences951    COMMON_CONTEXT_SEQ_RM_TYPE_FULL         = 2, // can seq_rm full sequences only952    COMMON_CONTEXT_SEQ_RM_TYPE_RS = 3, // can seq_rm partial sequences, bounded by n_rs_seq953};954 955// check if the llama_context can remove sequences956// note: clears the memory of the context957common_context_seq_rm_type common_context_can_seq_rm(llama_context * ctx);958 959struct common_memory {960    llama_context * ctx_tgt = nullptr;961    llama_context * ctx_dft = nullptr;962 963    void init(llama_context * ctx_tgt, llama_context * ctx_dft = nullptr);964 965    // aborts execution on failure966    void seq_rm (llama_seq_id seq_id, llama_pos p0, llama_pos p1) const;967    void seq_add(llama_seq_id seq_id, llama_pos p0, llama_pos p1, llama_pos delta) const;968    void seq_cp (llama_seq_id seq_id_src, llama_seq_id seq_id_dst, llama_pos p0, llama_pos p1) const;969};970 971//972// Batch utils973//974 975void common_batch_clear(struct llama_batch & batch);976 977void common_batch_add(978                 struct llama_batch & batch,979                        llama_token   id,980                          llama_pos   pos,981    const std::vector<llama_seq_id> & seq_ids,982                               bool   logits);983 984// decodes a single batch of tokens for a prompt and manages session tokens985//986// Note: We save state before the last token so that we can replay it to ensure987// compatibility with all memory types. Recurrent/hybrid models cannot remove988// tokens from memory, so this approach works across all model architectures.989bool common_prompt_batch_decode(990              struct llama_context * ctx,991    const std::vector<llama_token> & all_tokens,992                               int   n_new,993                               int & n_past,994                               int   n_batch,995                  std::string_view   state_path,996                              bool   save_state);997 998// replays the last token after loading state to regenerate logits999// used after loading session state to ensure the sampling context has valid logits1000bool common_replay_last_token(struct llama_context * ctx, llama_token last_token, int32_t pos);1001 1002//1003// Vocab utils1004//1005 1006// tokenizes a string into a vector of tokens1007// should work similar to Python's `tokenizer.encode`1008std::vector<llama_token> common_tokenize(1009  const struct llama_context * ctx,1010           const std::string & text,1011                        bool   add_special,1012                        bool   parse_special = false);1013 1014std::vector<llama_token> common_tokenize(1015    const struct llama_vocab * vocab,1016           const std::string & text,1017                        bool   add_special,1018                        bool   parse_special = false);1019 1020// tokenizes a token into a piece, optionally renders special/control tokens1021// should work similar to Python's `tokenizer.id_to_piece`1022std::string common_token_to_piece(1023        const struct llama_context * ctx,1024                       llama_token   token,1025                       bool          special = true);1026 1027std::string common_token_to_piece(1028          const struct llama_vocab * vocab,1029                       llama_token   token,1030                       bool          special = true);1031 1032// detokenizes a vector of tokens into a string1033// should work similar to Python's `tokenizer.decode`1034// optionally renders special/control tokens1035std::string common_detokenize(1036            const struct llama_context * ctx,1037        const std::vector<llama_token> & tokens,1038                                  bool   special = true);1039 1040std::string common_detokenize(1041              const struct llama_vocab * vocab,1042        const std::vector<llama_token> & tokens,1043                                  bool   special = true);1044 1045//1046// Embedding utils1047//1048 1049// TODO: replace embd_norm with an enum1050void common_embd_normalize(const float * inp, float * out, int n, int embd_norm);1051 1052float common_embd_similarity_cos(const float * embd1, const float * embd2, int n);1053 1054//1055// Control vector utils1056//1057 1058struct common_control_vector_data {1059    int n_embd;1060 1061    // stores data for layers [1, n_layer] where n_layer = data.size() / n_embd1062    std::vector<float> data;1063};1064 1065struct common_control_vector_load_info {1066    float strength;1067 1068    std::string fname;1069};1070 1071// Load control vectors, scale each by strength, and add them together.1072// On error, returns {-1, empty}1073common_control_vector_data common_control_vector_load(const std::vector<common_control_vector_load_info> & load_infos);1074 1075//1076// Split utils1077//1078 1079namespace {1080 1081const char * const LLM_KV_SPLIT_NO            = "split.no";1082const char * const LLM_KV_SPLIT_COUNT         = "split.count";1083const char * const LLM_KV_SPLIT_TENSORS_COUNT = "split.tensors.count";1084 1085}1086 1087//1088// MoE utils1089//1090 1091const char * const LLM_FFN_EXPS_REGEX = "\\.ffn_(up|down|gate|gate_up)_(ch|)exps";1092 1093inline std::string llm_ffn_exps_block_regex(int idx) {1094    return string_format("blk\\.%d%s", idx, LLM_FFN_EXPS_REGEX);1095}1096 1097inline llama_model_tensor_buft_override llm_ffn_exps_cpu_override() {1098    return { LLM_FFN_EXPS_REGEX, ggml_backend_cpu_buffer_type() };1099}1100 1101//1102// training utils1103//1104 1105ggml_opt_dataset_t common_opt_dataset_init(struct llama_context * ctx, const std::vector<llama_token> & tokens, int64_t stride);1106 1107// "adamw" or "sgd" (case insensitive)1108enum ggml_opt_optimizer_type common_opt_get_optimizer(const char *);1109 1110//1111// prompt utils1112//1113 1114struct common_prompt_checkpoint {1115    int64_t n_tokens;1116 1117    // (optional) id of the task that created the checkpoint1118    int id_task = -1;1119 1120    llama_pos pos_min;1121    llama_pos pos_max;1122 1123    std::vector<uint8_t> data_tgt;1124    std::vector<uint8_t> data_dft;1125 1126    // (optional) speculative-decoding implementation state stashed with the checkpoint1127    // (e.g. eagle3's deferred-boundary g_embd row)1128    std::vector<uint8_t> data_spec;1129 1130    size_t size() const;1131 1132    bool empty() const;1133    void clear();1134 1135    void update_pos(1136            int64_t n_tokens,1137            llama_pos pos_min,1138            llama_pos pos_max);1139 1140    void update_tgt(1141            llama_context * ctx,1142            llama_seq_id seq_id,1143            llama_state_seq_flags flags);1144 1145    void update_dft(1146            llama_context * ctx,1147            llama_seq_id seq_id,1148            llama_state_seq_flags flags);1149 1150    void load_tgt(1151            llama_context * ctx,1152            llama_seq_id seq_id,1153            llama_state_seq_flags flags) const;1154 1155    void load_dft(1156            llama_context * ctx,1157            llama_seq_id seq_id,1158            llama_state_seq_flags flags) const;1159 1160    void clear_tgt();1161    void clear_dft();1162};1163 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai