Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1// Various helper functions and utilities2 3#pragma once4 5#include "llama-cpp.h"6 7#include "ggml-opt.h"8#include "ggml.h"9#include "llama.h"10 11#include <set>12#include <sstream>13#include <string>14#include <string_view>15#include <vector>16#include <map>17#include <algorithm>18#include <fstream>19 20#if defined(_WIN32) && !defined(_WIN32_WINNT)21#define _WIN32_WINNT 0x0A0022#endif23 24#ifdef _WIN3225#define DIRECTORY_SEPARATOR '\\'26#else27#define DIRECTORY_SEPARATOR '/'28#endif // _WIN3229 30#define COM_DBG(fmt, ...) LOG_DBG("cmn %12.*s: " fmt, 12, __func__, __VA_ARGS__)31#define COM_TRC(fmt, ...) LOG_TRC("cmn %12.*s: " fmt, 12, __func__, __VA_ARGS__)32#define COM_INF(fmt, ...) LOG_INF("cmn %12.*s: " fmt, 12, __func__, __VA_ARGS__)33#define COM_WRN(fmt, ...) LOG_WRN("cmn %12.*s: " fmt, 12, __func__, __VA_ARGS__)34#define COM_ERR(fmt, ...) LOG_ERR("cmn %12.*s: " fmt, 12, __func__, __VA_ARGS__)35#define COM_CNT(fmt, ...) LOG_CNT("" fmt, __VA_ARGS__)36 37#define die(msg) do { fputs("error: " msg "\n", stderr); exit(1); } while (0)38#define die_fmt(fmt, ...) do { fprintf(stderr, "error: " fmt "\n", __VA_ARGS__); exit(1); } while (0)39 40struct common_time_meas {41 common_time_meas(int64_t & t_acc, bool disable = false);42 ~common_time_meas();43 44 const int64_t t_start_us;45 46 int64_t & t_acc;47};48 49struct common_adapter_lora_info {50 std::string path;51 float scale;52 53 std::string task_name;54 std::string prompt_prefix;55 56 struct llama_adapter_lora * ptr;57};58 59using llama_tokens = std::vector<llama_token>;60 61struct common_control_vector_load_info;62 63//64// CPU utils65//66 67struct common_cpu_params {68 int n_threads = -1;69 bool cpumask[GGML_MAX_N_THREADS] = {false}; // CPU affinity mask.70 bool mask_valid = false; // Default: any CPU71 enum ggml_sched_priority priority = GGML_SCHED_PRIO_NORMAL; // Scheduling prio : (0 - normal, 1 - medium, 2 - high, 3 - realtime)72 bool strict_cpu = false; // Use strict CPU placement73 uint32_t poll = 50; // Polling (busywait) level (0 - no polling, 100 - mostly polling)74};75 76int32_t common_cpu_get_num_physical_cores();77int32_t common_cpu_get_num_math();78 79//80// Common params81//82 83enum llama_example {84 LLAMA_EXAMPLE_BATCHED,85 LLAMA_EXAMPLE_DEBUG,86 LLAMA_EXAMPLE_COMMON,87 LLAMA_EXAMPLE_SPECULATIVE,88 LLAMA_EXAMPLE_COMPLETION,89 LLAMA_EXAMPLE_CLI,90 LLAMA_EXAMPLE_EMBEDDING,91 LLAMA_EXAMPLE_PERPLEXITY,92 LLAMA_EXAMPLE_RETRIEVAL,93 LLAMA_EXAMPLE_PASSKEY,94 LLAMA_EXAMPLE_IMATRIX,95 LLAMA_EXAMPLE_BENCH,96 LLAMA_EXAMPLE_SERVER,97 LLAMA_EXAMPLE_CVECTOR_GENERATOR,98 LLAMA_EXAMPLE_EXPORT_LORA,99 LLAMA_EXAMPLE_MTMD,100 LLAMA_EXAMPLE_LOOKUP,101 LLAMA_EXAMPLE_PARALLEL,102 LLAMA_EXAMPLE_TTS,103 LLAMA_EXAMPLE_DIFFUSION,104 LLAMA_EXAMPLE_FINETUNE,105 LLAMA_EXAMPLE_FIT_PARAMS,106 LLAMA_EXAMPLE_RESULTS,107 LLAMA_EXAMPLE_EXPORT_GRAPH_OPS,108 LLAMA_EXAMPLE_DOWNLOAD,109 LLAMA_EXAMPLE_TOKENIZE,110 111 LLAMA_EXAMPLE_COUNT,112};113 114enum common_sampler_type {115 COMMON_SAMPLER_TYPE_NONE = 0,116 COMMON_SAMPLER_TYPE_DRY = 1,117 COMMON_SAMPLER_TYPE_TOP_K = 2,118 COMMON_SAMPLER_TYPE_TOP_P = 3,119 COMMON_SAMPLER_TYPE_MIN_P = 4,120 //COMMON_SAMPLER_TYPE_TFS_Z = 5,121 COMMON_SAMPLER_TYPE_TYPICAL_P = 6,122 COMMON_SAMPLER_TYPE_TEMPERATURE = 7,123 COMMON_SAMPLER_TYPE_XTC = 8,124 COMMON_SAMPLER_TYPE_INFILL = 9,125 COMMON_SAMPLER_TYPE_PENALTIES = 10,126 COMMON_SAMPLER_TYPE_TOP_N_SIGMA = 11,127 COMMON_SAMPLER_TYPE_ADAPTIVE_P = 12,128};129 130// dimensionality reduction methods, used by cvector-generator131enum dimre_method {132 DIMRE_METHOD_PCA,133 DIMRE_METHOD_MEAN,134};135 136enum common_conversation_mode {137 COMMON_CONVERSATION_MODE_DISABLED = 0,138 COMMON_CONVERSATION_MODE_ENABLED = 1,139 COMMON_CONVERSATION_MODE_AUTO = 2,140};141 142enum common_grammar_trigger_type {143 COMMON_GRAMMAR_TRIGGER_TYPE_TOKEN,144 COMMON_GRAMMAR_TRIGGER_TYPE_WORD,145 COMMON_GRAMMAR_TRIGGER_TYPE_PATTERN,146 COMMON_GRAMMAR_TRIGGER_TYPE_PATTERN_FULL,147};148 149struct common_grammar_trigger {150 common_grammar_trigger_type type;151 std::string value;152 llama_token token = LLAMA_TOKEN_NULL;153};154 155enum common_params_sampling_config : uint64_t {156 COMMON_PARAMS_SAMPLING_CONFIG_SAMPLERS = 1 << 0,157 COMMON_PARAMS_SAMPLING_CONFIG_TOP_K = 1 << 1,158 COMMON_PARAMS_SAMPLING_CONFIG_TOP_P = 1 << 2,159 COMMON_PARAMS_SAMPLING_CONFIG_MIN_P = 1 << 3,160 COMMON_PARAMS_SAMPLING_CONFIG_XTC_PROBABILITY = 1 << 4,161 COMMON_PARAMS_SAMPLING_CONFIG_XTC_THRESHOLD = 1 << 5,162 COMMON_PARAMS_SAMPLING_CONFIG_TEMP = 1 << 6,163 COMMON_PARAMS_SAMPLING_CONFIG_PENALTY_LAST_N = 1 << 7,164 COMMON_PARAMS_SAMPLING_CONFIG_PENALTY_REPEAT = 1 << 8,165 COMMON_PARAMS_SAMPLING_CONFIG_MIROSTAT = 1 << 9,166 COMMON_PARAMS_SAMPLING_CONFIG_MIROSTAT_TAU = 1 << 10,167 COMMON_PARAMS_SAMPLING_CONFIG_MIROSTAT_ETA = 1 << 11,168};169 170enum common_speculative_type {171 COMMON_SPECULATIVE_TYPE_NONE, // no speculative decoding172 COMMON_SPECULATIVE_TYPE_DRAFT_SIMPLE, // standalone draft model speculative decoding173 COMMON_SPECULATIVE_TYPE_DRAFT_EAGLE3, // Eagle3 speculative decoding174 COMMON_SPECULATIVE_TYPE_DRAFT_MTP, // Multi-token prediction175 COMMON_SPECULATIVE_TYPE_DRAFT_DFLASH, // DFlash speculative decoding176 COMMON_SPECULATIVE_TYPE_DRAFT_DSPARK, // DSpark speculative decoding (DFlash + Markov head)177 COMMON_SPECULATIVE_TYPE_NGRAM_SIMPLE, // simple self-speculative decoding based on n-grams178 COMMON_SPECULATIVE_TYPE_NGRAM_MAP_K, // self-speculative decoding with n-gram keys only179 COMMON_SPECULATIVE_TYPE_NGRAM_MAP_K4V, // self-speculative decoding with n-gram keys and 4 m-gram values180 COMMON_SPECULATIVE_TYPE_NGRAM_MOD,181 COMMON_SPECULATIVE_TYPE_NGRAM_CACHE, // self-speculative decoding with 3-level n-gram cache182 COMMON_SPECULATIVE_TYPE_COUNT // number of types, unknown type183};184 185// Grammar type enumeration186enum common_grammar_type {187 COMMON_GRAMMAR_TYPE_NONE, // no grammar set188 COMMON_GRAMMAR_TYPE_USER, // user-provided GBNF (--grammar / "grammar" API field)189 COMMON_GRAMMAR_TYPE_OUTPUT_FORMAT, // auto-generated from JSON schema (--json-schema / "json_schema" API field)190 COMMON_GRAMMAR_TYPE_TOOL_CALLS, // auto-generated by chat template parser for function calling191};192 193// Grammar variant struct with type and grammar string194struct common_grammar {195 common_grammar_type type = COMMON_GRAMMAR_TYPE_NONE;196 std::string grammar;197 198 // Default constructor - no grammar199 common_grammar() = default;200 201 // Constructor with type and grammar string202 common_grammar(common_grammar_type t, std::string g) : type(t), grammar(std::move(g)) {203 GGML_ASSERT(type != COMMON_GRAMMAR_TYPE_NONE || !grammar.empty());204 }205 206 // Check if a grammar is set207 bool empty() const { return type == COMMON_GRAMMAR_TYPE_NONE || grammar.empty(); }208};209 210// Returns the raw grammar string, or empty string if no grammar is set.211inline const std::string & common_grammar_value(const common_grammar & g) {212 return g.grammar;213}214 215// Returns true when the generation_prompt should be prefilled into the grammar sampler.216// Only output-format and tool-call grammars need prefill; user-supplied grammars must not be prefilled.217inline bool common_grammar_needs_prefill(const common_grammar & g) {218 return g.type == COMMON_GRAMMAR_TYPE_OUTPUT_FORMAT219 || g.type == COMMON_GRAMMAR_TYPE_TOOL_CALLS;220}221 222// sampling parameters223struct common_params_sampling {224 uint32_t seed = LLAMA_DEFAULT_SEED; // the seed used to initialize llama_sampler225 226 int32_t n_prev = 64; // number of previous tokens to remember227 int32_t n_probs = 0; // if greater than 0, output the probabilities of top n_probs tokens.228 int32_t min_keep = 0; // 0 = disabled, otherwise samplers should return at least min_keep tokens229 int32_t top_k = 40; // <= 0 to use vocab size230 float top_p = 0.95f; // 1.0 = disabled231 float min_p = 0.05f; // 0.0 = disabled232 float xtc_probability = 0.00f; // 0.0 = disabled233 float xtc_threshold = 0.10f; // > 0.5 disables XTC234 float typ_p = 1.00f; // typical_p, 1.0 = disabled235 float temp = 0.80f; // <= 0.0 to sample greedily, 0.0 to not output probabilities236 float dynatemp_range = 0.00f; // 0.0 = disabled237 float dynatemp_exponent = 1.00f; // controls how entropy maps to temperature in dynamic temperature sampler238 int32_t penalty_last_n = 64; // last n tokens to penalize (0 = disable penalty)239 float penalty_repeat = 1.00f; // 1.0 = disabled240 float penalty_freq = 0.00f; // 0.0 = disabled241 float penalty_present = 0.00f; // 0.0 = disabled242 float dry_multiplier = 0.0f; // 0.0 = disabled; DRY repetition penalty for tokens extending repetition:243 float dry_base = 1.75f; // 0.0 = disabled; multiplier * base ^ (length of sequence before token - allowed length)244 int32_t dry_allowed_length = 2; // tokens extending repetitions beyond this receive penalty245 int32_t dry_penalty_last_n = 64; // how many tokens to scan for repetitions (0 = disable penalty)246 float adaptive_target = -1.0f; // select tokens near this probability (valid range 0.0 to 1.0; negative = disabled)247 float adaptive_decay = 0.90f; // EMA decay for adaptation; history ≈ 1/(1-decay) tokens (0.0 - 0.99)248 int32_t mirostat = 0; // 0 = disabled, 1 = mirostat, 2 = mirostat 2.0249 float top_n_sigma = -1.00f; // -1.0 = disabled250 float mirostat_tau = 5.00f; // target entropy251 float mirostat_eta = 0.10f; // learning rate252 bool ignore_eos = false;253 bool no_perf = false; // disable performance metrics254 bool timing_per_token = false;255 256 uint64_t user_sampling_config = 0; // bitfield to track user-specified samplers257 258 std::vector<std::string> dry_sequence_breakers = {"\n", ":", "\"", "*"}; // default sequence breakers for DRY259 260 std::vector<enum common_sampler_type> samplers = {261 COMMON_SAMPLER_TYPE_PENALTIES,262 COMMON_SAMPLER_TYPE_DRY,263 COMMON_SAMPLER_TYPE_TOP_N_SIGMA,264 COMMON_SAMPLER_TYPE_TOP_K,265 COMMON_SAMPLER_TYPE_TYPICAL_P,266 COMMON_SAMPLER_TYPE_TOP_P,267 COMMON_SAMPLER_TYPE_MIN_P,268 COMMON_SAMPLER_TYPE_XTC,269 COMMON_SAMPLER_TYPE_TEMPERATURE,270 };271 272 common_grammar grammar; // optional grammar constraint (user / output-format / tool-calls)273 bool grammar_lazy = false;274 std::vector<common_grammar_trigger> grammar_triggers; // optional triggers (for lazy grammars)275 std::set<llama_token> preserved_tokens;276 277 std::vector<llama_logit_bias> logit_bias; // logit biases to apply278 std::vector<llama_logit_bias> logit_bias_eog; // pre-calculated logit biases for EOG tokens279 280 // The assistant generation prompt already prefilled into the prompt.281 // Fed to the grammar sampler (to advance past pre-existing tokens) and used282 // to determine the reasoning budget sampler's initial state.283 // Only applied when the grammar is of output-format or tool-calls type.284 std::string generation_prompt;285 286 // reasoning budget sampler parameters287 // these are populated by the server/CLI based on chat template params288 int32_t reasoning_budget_tokens = -1; // -1 = disabled, >= 0 = token budget289 std::vector<llama_token> reasoning_budget_start; // start tag token sequence290 std::vector<llama_tokens> reasoning_budget_end; // end tag token sequences; the first tag is used as the forcing sequence291 std::vector<llama_token> reasoning_budget_forced; // forced sequence (message + first end tag)292 std::string reasoning_budget_message; // message injected before end tag when budget exhausted293 bool reasoning_control = false; // create the budget sampler on demand so reasoning can be ended at runtime294 295 bool backend_sampling = false;296 297 // print the parameters into a string298 std::string print() const;299};300 301struct common_params_model {302 std::string path = ""; // model local path303 std::string url = ""; // model url to download304 std::string hf_repo = ""; // HF repo305 std::string hf_file = ""; // HF file306 std::string docker_repo = ""; // Docker repo307 308 std::string get_name() const {309 if (!hf_repo.empty()) {310 return hf_repo;311 }312 if (!docker_repo.empty()) {313 return docker_repo;314 }315 return path;316 }317 318 bool empty() const {319 return get_name().empty();320 }321};322 323// draft-model-based speculative decoding parameters324struct common_params_speculative_draft {325 int32_t n_max = 8; // maximum number of tokens to draft during speculative decoding (optimized for code)326 int32_t n_min = 0; // minimum number of draft tokens to use for speculative decoding327 328 float p_split = 0.1f; // speculative decoding split probability329 float p_min = 0.75f; // minimum speculative decoding probability (greedy)330 331 bool backend_sampling = true; // offload draft sampling to the backend (default: on)332 333 common_params_model mparams;334 335 llama_context * ctx_tgt = nullptr;336 llama_context * ctx_dft = nullptr;337 338 int32_t n_gpu_layers = -1; // number of layers to store in VRAM for the draft model (-1 - use default)339 340 ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K341 ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V342 343 common_cpu_params cpuparams;344 common_cpu_params cpuparams_batch;345 346 std::vector<ggml_backend_dev_t> devices; // devices to use for offloading347 348 std::vector<llama_model_tensor_buft_override> tensor_buft_overrides;349};350 351struct common_params_speculative_ngram_mod {352 int32_t n_match = 24;353 354 int32_t n_max = 64;355 int32_t n_min = 48;356};357 358struct common_params_speculative_ngram_map {359 uint16_t size_n = 12; // ngram size for lookup360 uint16_t size_m = 48; // mgram size for speculative tokens361 uint16_t min_hits = 1; // minimum hits at ngram/mgram lookup for mgram to be proposed362};363 364struct common_params_speculative_ngram_cache {365 std::string lookup_cache_static; // path of static ngram cache file for lookup decoding366 std::string lookup_cache_dynamic; // path of dynamic ngram cache file for lookup decoding367};368 369struct common_params_speculative {370 std::vector<enum common_speculative_type> types = { COMMON_SPECULATIVE_TYPE_NONE };371 372 // used by Simple, MTP, Eagle3, etc. - all methods that require some kind of draft model373 common_params_speculative_draft draft;374 375 common_params_speculative_ngram_mod ngram_mod;376 common_params_speculative_ngram_map ngram_simple;377 common_params_speculative_ngram_map ngram_map_k;378 common_params_speculative_ngram_map ngram_map_k4v;379 380 common_params_speculative_ngram_cache ngram_cache;381 382 bool has_dft() const {383 return !draft.mparams.empty();384 }385 386 uint32_t need_n_rs_seq() const {387 bool needs_rs_seq = std::any_of(types.begin(), types.end(), [&](auto t) {388 return t == COMMON_SPECULATIVE_TYPE_DRAFT_MTP || t == COMMON_SPECULATIVE_TYPE_DRAFT_EAGLE3 || t == COMMON_SPECULATIVE_TYPE_DRAFT_DFLASH || t == COMMON_SPECULATIVE_TYPE_DRAFT_DSPARK;389 });390 391 return needs_rs_seq ? draft.n_max : 0u;392 }393};394 395struct common_params_diffusion {396 int32_t steps = 128;397 bool visual_mode = false;398 399 float eps = 0; // epsilon for timesteps400 int32_t block_length = 0; // block length for generation401 402 int32_t algorithm = 4; // default algorithm: low-confidence403 float alg_temp = 0.0f; // algorithm temperature404 405 float cfg_scale = 0; // classifier-free guidance scale406 bool add_gumbel_noise = false; // add gumbel noise to the logits if temp > 0.0407};408 409// reasoning API response format (not to be confused as chat template's reasoning format)410// only used by server411enum common_reasoning_format {412 COMMON_REASONING_FORMAT_NONE,413 COMMON_REASONING_FORMAT_AUTO, // Same as deepseek, using `message.reasoning_content`414 COMMON_REASONING_FORMAT_DEEPSEEK_LEGACY, // Extract thinking tag contents and return as `message.reasoning_content`, or leave inline in <think> tags in stream mode415 COMMON_REASONING_FORMAT_DEEPSEEK, // Extract thinking tag contents and return as `message.reasoning_content`, including in streaming deltas.416 // do not extend this enum unless you absolutely have to417 // in most cases, use COMMON_REASONING_FORMAT_AUTO418 // see: https://github.com/ggml-org/llama.cpp/pull/15408419};420 421 422struct lr_opt {423 float lr0 = 1e-5; // learning rate at first epoch424 float lr_min = -1;425 float decay_epochs = -1; // if >0, the learning rate starts at lr0 and decays to lr_min after this many epochs426 float scale_epoch = 0;427 float wd = 0;428 unsigned epochs = 2;429 430 unsigned epoch; // set by optimizer outer (epochs) loop431 // learning rate decay - constant LR per epoch only for now432 float get_lr(float e) const;433 float get_lr() const { return get_lr(epoch); }434 // must call after arg parse, before get_lr435 void init();436};437 438struct ggml_opt_optimizer_params common_opt_lr_pars(void * userdata);439 440struct common_params {441 int32_t n_predict = -1; // max. number of new tokens to predict, -1 == no limit442 int32_t n_ctx = 0; // context size, 0 == context the model was trained with443 int32_t n_batch = 2048; // logical batch size for prompt processing (must be >=32 to use BLAS)444 int32_t n_ubatch = 512; // physical batch size for prompt processing (must be >=32 to use BLAS)445 int32_t n_keep = 0; // number of tokens to keep from initial prompt446 int32_t n_chunks = -1; // max number of chunks to process (-1 = unlimited)447 int32_t n_parallel = 1; // number of parallel sequences to decode448 int32_t n_sequences = 1; // number of sequences to decode449 int32_t n_outputs_max = 0; // max outputs in a batch (0 = n_batch)450 int32_t n_outputs_max_per_seq = 1; // max outputs per sequence451 int32_t grp_attn_n = 1; // group-attention factor452 int32_t grp_attn_w = 512; // group-attention width453 int32_t n_print = -1; // print token count every n tokens (-1 = disabled)454 float rope_freq_base = 0.0f; // RoPE base frequency455 float rope_freq_scale = 0.0f; // RoPE frequency scaling factor456 float yarn_ext_factor = -1.0f; // YaRN extrapolation mix factor457 float yarn_attn_factor = -1.0f; // YaRN magnitude scaling factor458 float yarn_beta_fast = -1.0f; // YaRN low correction dim459 float yarn_beta_slow = -1.0f; // YaRN high correction dim460 int32_t yarn_orig_ctx = 0; // YaRN original context length461 462 // offload params463 std::vector<ggml_backend_dev_t> devices; // devices to use for offloading464 465 int32_t n_gpu_layers = -1; // number of layers to store in VRAM, -1 is auto, <= -2 is all466 int32_t main_gpu = 0; // the GPU that is used for scratch and small tensors467 float tensor_split[128] = {0}; // how split tensors should be distributed across GPUs468 bool fit_params = true; // whether to fit unset model/context parameters to free device memory469 bool fit_params_print = false; // print the estimated required memory to run the model470 int32_t fit_params_min_ctx = 4096; // minimum context size to set when trying to reduce memory use471 472 // margin per device in bytes for fitting parameters to free memory:473 std::vector<size_t> fit_params_target = std::vector<size_t>(llama_max_devices(), 1024 * 1024*1024);474 475 enum llama_split_mode split_mode = LLAMA_SPLIT_MODE_LAYER; // how to split the model across GPUs476 enum llama_load_mode load_mode = LLAMA_LOAD_MODE_MMAP; // how to load the model477 478 common_cpu_params cpuparams;479 common_cpu_params cpuparams_batch;480 481 ggml_backend_sched_eval_callback cb_eval = nullptr;482 void * cb_eval_user_data = nullptr;483 484 ggml_numa_strategy numa = GGML_NUMA_STRATEGY_DISABLED;485 486 enum llama_rope_scaling_type rope_scaling_type = LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED;487 enum llama_pooling_type pooling_type = LLAMA_POOLING_TYPE_UNSPECIFIED; // pooling type for embeddings488 enum llama_attention_type attention_type = LLAMA_ATTENTION_TYPE_UNSPECIFIED; // attention type for embeddings489 enum llama_flash_attn_type flash_attn_type = LLAMA_FLASH_ATTN_TYPE_AUTO; // whether to use Flash Attention490 491 struct common_params_sampling sampling;492 struct common_params_speculative speculative;493 struct common_params_diffusion diffusion;494 495 struct common_params_model model;496 497 std::set<std::string> model_alias; // model aliases // NOLINT498 std::set<std::string> model_tags; // model tags (informational, not used for routing) // NOLINT499 std::string hf_token = ""; // HF token (aka bearer token) // NOLINT500 std::string prompt = ""; // NOLINT501 std::string system_prompt = ""; // NOLINT502 std::string prompt_file = ""; // store the external prompt file name // NOLINT503 std::string path_prompt_cache = ""; // path to file for saving/loading prompt eval state // NOLINT504 std::string input_prefix = ""; // string to prefix user inputs with // NOLINT505 std::string input_suffix = ""; // string to suffix user inputs with // NOLINT506 std::string logits_file = ""; // file for saving *all* logits // NOLINT507 std::string path_prompts_log_dir = ""; // directory with logged prompts // NOLINT508 509 // llama-debug specific options510 std::string logits_output_dir = "data"; // directory for saving logits output files // NOLINT511 bool save_logits = false; // whether to save logits to files // NOLINT512 std::vector<std::string> tensor_filter; // filter tensor names for debug output (regex) // NOLINT513 514 std::vector<std::string> in_files; // all input files515 std::vector<std::string> antiprompt; // strings upon which more user input is prompted (a.k.a. reverse prompts)516 std::vector<llama_model_kv_override> kv_overrides;517 std::vector<llama_model_tensor_buft_override> tensor_buft_overrides;518 519 bool lora_init_without_apply = false; // only load lora to memory, but do not apply it to ctx (user can manually apply lora later using llama_adapter_lora_apply)520 std::vector<common_adapter_lora_info> lora_adapters; // lora adapter path with user defined scale521 522 std::vector<common_control_vector_load_info> control_vectors; // control vector with user defined scale523 524 int32_t verbosity = 3; // LOG_LEVEL_INFO525 int32_t control_vector_layer_start = -1; // layer range for control vector526 int32_t control_vector_layer_end = -1; // layer range for control vector527 bool offline = false;528 529 int32_t ppl_stride = 0; // stride for perplexity calculations. If left at 0, the pre-existing approach will be used.530 int32_t ppl_output_type = 0; // = 0 -> ppl output is as usual, = 1 -> ppl output is num_tokens, ppl, one per line531 // (which is more convenient to use for plotting)532 //533 bool hellaswag = false; // compute HellaSwag score over random tasks from datafile supplied in prompt534 size_t hellaswag_tasks = 400; // number of tasks to use when computing the HellaSwag score535 536 bool winogrande = false; // compute Winogrande score over random tasks from datafile supplied in prompt537 size_t winogrande_tasks = 0; // number of tasks to use when computing the Winogrande score. If 0, all tasks will be computed538 539 bool multiple_choice = false; // compute TruthfulQA score over random tasks from datafile supplied in prompt540 size_t multiple_choice_tasks = 0; // number of tasks to use when computing the TruthfulQA score. If 0, all tasks will be computed541 542 bool kl_divergence = false; // compute KL divergence543 544 bool check = false; // check rather than generate results for llama-results545 546 bool usage = false; // print usage547 bool completion = false; // print source-able completion script548 bool use_color = false; // use color to distinguish generations and inputs549 bool special = false; // enable special token output550 bool interactive = false; // interactive mode551 bool interactive_first = false; // wait for user input immediately552 bool prompt_cache_all = false; // save user input and generations to prompt cache553 bool prompt_cache_ro = false; // open the prompt cache read-only and do not update it554 555 bool escape = true; // escape "\n", "\r", "\t", "\'", "\"", and "\\"556 bool multiline_input = false; // reverse the usage of `\`557 bool simple_io = false; // improves compatibility with subprocesses and limited consoles558 bool cont_batching = true; // insert new sequences for decoding on-the-fly559 bool no_perf = false; // disable performance metrics560 bool show_timings = true; // show timing information on CLI561 bool ctx_shift = false; // context shift on infinite text generation562 bool swa_full = false; // use full-size SWA cache (https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)563 bool kv_unified = false; // enable unified KV cache564 565 bool input_prefix_bos = false; // prefix BOS to user inputs, preceding input_prefix566 bool verbose_prompt = false; // print prompt tokens before generation567 bool display_prompt = true; // print prompt before generation568 bool no_kv_offload = false; // disable KV offloading569 bool warmup = true; // warmup run570 bool check_tensors = false; // validate tensor data571 bool no_op_offload = false; // globally disable offload host tensor operations to device572 bool no_extra_bufts = false; // disable extra buffer types (used for weight repacking)573 bool no_host = false; // bypass host buffer allowing extra buffers to be used574 575 bool single_turn = false; // single turn chat conversation576 577 ggml_type cache_type_k = GGML_TYPE_F16; // KV cache data type for the K578 ggml_type cache_type_v = GGML_TYPE_F16; // KV cache data type for the V579 580 common_conversation_mode conversation_mode = COMMON_CONVERSATION_MODE_AUTO;581 582 // multimodal models (see tools/mtmd)583 struct common_params_model mmproj;584 bool mmproj_use_gpu = true; // use GPU for multimodal model585 bool no_mmproj = false; // explicitly disable multimodal model586 std::vector<std::string> image; // path to image file(s) ; TODO: change the name to "media"587 int image_min_tokens = -1;588 int image_max_tokens = -1;589 int mtmd_batch_max_tokens = 1024;590 591 // finetune592 struct lr_opt lr;593 enum ggml_opt_optimizer_type optimizer = GGML_OPT_OPTIMIZER_TYPE_ADAMW;594 float val_split = 0.05f; // fraction of the data used for the validation set595 596 // embedding597 bool embedding = false; // get only sentence embedding598 int32_t embd_normalize = 2; // normalisation for embeddings (-1=none, 0=max absolute int16, 1=taxicab, 2=euclidean, >2=p-norm)599 std::string embd_out = ""; // empty = default, "array" = [[],[]...], "json" = openai style, "json+" = same "json" + cosine similarity matrix600 std::string embd_sep = "\n"; // separator of embeddings601 std::string cls_sep = "\t"; // separator of classification sequences602 603 // server params604 int32_t port = 8080; // server listens on this network port605 bool reuse_port = false; // allow multiple sockets to bind to the same port606 int32_t timeout_read = 3600; // http read timeout in seconds607 int32_t timeout_write = timeout_read; // http write timeout in seconds608 int32_t sse_ping_interval = 30; // SSE ping interval in seconds609 int32_t n_threads_http = -1; // number of threads to process HTTP requests (TODO: support threadpool)610 int32_t n_cache_reuse = 0; // min chunk size to reuse from the cache via KV shifting611 bool cache_prompt = true; // whether to enable prompt caching612 bool cache_idle_slots = true; // save and clear idle slots upon starting a new task613 int32_t n_ctx_checkpoints = 32; // max number of context checkpoints per slot614 int32_t checkpoint_min_step = 8192; // minimum spacing between context checkpoints615 int32_t cache_ram_mib = 8192; // -1 = no limit, 0 - disable, 1 = 1 MiB, etc.616 617 std::string hostname = "127.0.0.1";618 std::string public_path = ""; // NOLINT619 std::string api_prefix = ""; // NOLINT620 std::string chat_template = ""; // NOLINT621 bool use_jinja = true; // NOLINT622 623 // server CORS params624 std::string cors_origins = "*";625 std::string cors_methods = "GET, POST, DELETE, OPTIONS";626 std::string cors_headers = "*";627 bool cors_credentials = true;628 bool cors_origins_explicit = false; // for --agent option629 630 bool enable_chat_template = true;631 bool force_pure_content_parser = false;632 common_reasoning_format reasoning_format = COMMON_REASONING_FORMAT_DEEPSEEK;633 int enable_reasoning = -1; // -1 = auto, 0 = disable, 1 = enable634 bool prefill_assistant = true; // if true, any trailing assistant message will be prefilled into the response635 int sleep_idle_seconds = -1; // if >0, server will sleep after this many seconds of idle time636 637 std::vector<std::string> api_keys;638 639 std::string ssl_file_key = ""; // NOLINT640 std::string ssl_file_cert = ""; // NOLINT641 642 std::map<std::string, std::string> default_template_kwargs;643 644 // CLI params645 std::string server_base; // if set, connect to this server instead of starting a new one646 647 // UI configs648 bool ui = true;649 bool ui_mcp_proxy = true;650 std::string ui_config_json;651 652 // "advanced" endpoints are disabled by default for better security653 bool endpoint_slots = true;654 bool endpoint_props = false; // only control POST requests, not GET655 bool endpoint_metrics = false;656 657 // enable built-in tools (enabled by default)658 std::vector<std::string> server_tools = {"all"};659 std::string server_tools_runtime;660 661 // MCP server configs (Cursor-compatible JSON)662 std::string mcp_servers_config; // path to JSON file with MCP server definitions663 std::string mcp_servers_json; // inline JSON with MCP server definitions664 665 // router server configs666 std::string models_dir = ""; // directory containing models for the router server667 std::string models_preset = ""; // directory containing model presets for the router server668 int models_max = 4; // maximum number of models to load simultaneously669 bool models_autoload = true; // automatically load models when requested via the router server670 std::string models_preset_hf = ""; // show a warning about remote presets on router loaded (if not empty)671 672 bool log_json = false;673 674 std::string slot_save_path;675 std::string media_path; // path to directory for loading media files676 677 float slot_prompt_similarity = 0.1f;678 679 // batched-bench params680 bool is_pp_shared = false;681 bool is_tg_separate = false;682 683 std::vector<int32_t> n_pp;684 std::vector<int32_t> n_tg;685 std::vector<int32_t> n_pl;686 687 // retrieval params688 std::vector<std::string> context_files; // context files to embed689 690 int32_t chunk_size = 64; // chunk size for context embedding691 692 std::string chunk_separator = "\n"; // chunk separator for context embedding693 694 // passkey params695 int32_t n_junk = 250; // number of times to repeat the junk text696 int32_t i_pos = -1; // position of the passkey in the junk text697 698 // imatrix params699 int32_t n_out_freq = 10; // output the imatrix every n_out_freq iterations700 int32_t n_save_freq = 0; // save the imatrix every n_save_freq iterations701 int32_t i_chunk = 0; // start processing from this chunk702 int8_t imat_dat = 0; // whether the legacy imatrix.dat format should be output (gguf <= 0 < dat)703 704 bool process_output = false; // collect data for the output tensor705 bool compute_ppl = true; // whether to compute perplexity706 bool show_statistics = false; // show imatrix statistics per tensor707 bool parse_special = false; // whether to parse special tokens during imatrix tokenization708 709 // cvector-generator params710 int n_pca_batch = 100;711 int n_pca_iterations = 1000;712 dimre_method cvector_dimre_method = DIMRE_METHOD_PCA;713 std::string cvector_positive_file = "tools/cvector-generator/positive.txt";714 std::string cvector_negative_file = "tools/cvector-generator/negative.txt";715 716 bool spm_infill = false; // suffix/prefix/middle pattern for infill717 718 // batched-bench params719 bool batched_bench_output_jsonl = false;720 721 // tokenize params722 bool tokenize_ids = false; // if true, only print the token IDs723 bool tokenize_stdin = false; // if true, read the prompt from stdin724 bool tokenize_no_bos = false; // if true, do not add the BOS token725 bool tokenize_show_count = false; // if true, print the total token count726 727 // common params728 std::string out_file; // output filename for all example programs729 // optional callback for model loading progress and cancellation:730 // called with a progress value between 0.0 and 1.0.731 // return false from callback to abort model loading or true to continue732 llama_progress_callback load_progress_callback = NULL;733 void * load_progress_callback_user_data = NULL;734 bool no_alloc = false; // Don't allocate model buffers735 736 // TTS params737 std::string tts_lang = "";738 std::string tts_speaker_file = "";739 740 bool is_gen_docs = false; // whether we are running inside llama-gen-docs741};742 743// call once at the start of a program if it uses libcommon744// initializes the logging system and prints info about the build745void common_init();746 747void common_params_print_info(const common_params & params, bool print_devices = true);748std::string common_params_get_system_info(const common_params & params);749 750bool parse_cpu_range(const std::string & range, bool(&boolmask)[GGML_MAX_N_THREADS]);751bool parse_cpu_mask(const std::string & mask, bool(&boolmask)[GGML_MAX_N_THREADS]);752void postprocess_cpu_params(common_cpu_params & cpuparams, const common_cpu_params * role_model = nullptr);753bool set_process_priority(enum ggml_sched_priority prio);754 755//756// String utils757//758 759#ifdef __GNUC__760# if defined(__MINGW32__) && !defined(__clang__)761# define LLAMA_COMMON_ATTRIBUTE_FORMAT(...) __attribute__((format(gnu_printf, __VA_ARGS__)))762# else763# define LLAMA_COMMON_ATTRIBUTE_FORMAT(...) __attribute__((format(printf, __VA_ARGS__)))764# endif765#else766# define LLAMA_COMMON_ATTRIBUTE_FORMAT(...)767#endif768 769LLAMA_COMMON_ATTRIBUTE_FORMAT(1, 2)770std::string string_format(const char * fmt, ...);771 772std::string string_strip(const std::string & str);773std::string string_get_sortable_timestamp();774std::string string_lcs(std::string_view a, std::string_view b);775 776std::string string_join(const std::vector<std::string> & values, const std::string & separator);777std::vector<std::string> string_split(const std::string & str, const std::string & delimiter);778std::string string_repeat(const std::string & str, size_t n);779 780void string_replace_all(std::string & s, const std::string & search, const std::string & replace);781 782std::string regex_escape(const std::string & s);783 784template<class T>785static std::vector<T> string_split(const std::string & str, char delim) {786 static_assert(!std::is_same<T, std::string>::value, "Please use the specialized version for std::string");787 std::vector<T> values;788 std::istringstream str_stream(str);789 std::string token;790 while (std::getline(str_stream, token, delim)) {791 T value;792 std::istringstream token_stream(token);793 token_stream >> value;794 values.push_back(value);795 }796 return values;797}798 799template<>800inline std::vector<std::string> string_split<std::string>(const std::string & str, char delim)801{802 std::vector<std::string> parts;803 size_t begin_pos = 0;804 size_t delim_pos = str.find(delim);805 while (delim_pos != std::string::npos) {806 std::string part = str.substr(begin_pos, delim_pos - begin_pos);807 parts.emplace_back(part);808 begin_pos = delim_pos + 1;809 delim_pos = str.find(delim, begin_pos);810 }811 parts.emplace_back(str.substr(begin_pos));812 return parts;813}814 815// remove when moving to c++20816inline bool string_starts_with(std::string_view str, std::string_view prefix) {817 return str.size() >= prefix.size() &&818 str.compare(0, prefix.size(), prefix) == 0;819}820 821// remove when moving to c++20822inline bool string_starts_with(std::string_view str, char prefix) {823 return !str.empty() && str.front() == prefix;824}825 826// remove when moving to c++20827inline bool string_ends_with(std::string_view str, std::string_view suffix) {828 return str.size() >= suffix.size() &&829 str.compare(str.size() - suffix.size(), suffix.size(), suffix) == 0;830}831 832inline bool string_remove_suffix(std::string & str, std::string_view suffix) {833 if (string_ends_with(str, suffix)) {834 str.resize(str.size() - suffix.size());835 return true;836 }837 return false;838}839 840inline size_t string_find_partial_stop(std::string_view str, std::string_view stop) {841 if (!str.empty() && !stop.empty()) {842 const size_t max_len = std::min(str.size(), stop.size());843 const char last_char = str.back();844 for (size_t len = max_len; len > 0; --len) {845 if (stop[len - 1] == last_char) {846 if (string_ends_with(str, stop.substr(0, len))) {847 return str.size() - len;848 }849 }850 }851 }852 return std::string::npos;853}854 855bool string_parse_kv_override(const char * data, std::vector<llama_model_kv_override> & overrides);856void string_process_escapes(std::string & input);857 858std::string string_from(bool value);859std::string string_from(const std::vector<int> & values);860std::string string_from(const struct llama_context * ctx, const std::vector<llama_token> & tokens);861std::string string_from(const struct llama_context * ctx, const struct llama_batch & batch);862 863bool glob_match(const std::string & pattern, const std::string & str);864 865//866// Environment utils867//868 869// portable environment access, an unset variable reads as an empty string870// and setting an empty value unsets the variable871std::string common_get_env(const std::string & name);872void common_set_env(const std::string & name, const std::string & value);873 874//875// Filesystem utils876//877 878bool fs_validate_filename(const std::string & filename, bool allow_subdirs = false);879bool fs_create_directory_with_parents(const std::string & path);880bool fs_is_directory(const std::string & path);881 882std::string fs_get_cache_directory();883std::string fs_get_cache_file(const std::string & filename);884 885struct common_file_info {886 std::string path;887 std::string name;888 size_t size = 0; // in bytes889 bool is_dir = false;890};891std::vector<common_file_info> fs_list(const std::string & path, bool include_directories);892 893// fs open, also handle UTF8 on Windows894std::ifstream fs_open_ifstream(const std::string & fname, std::ios_base::openmode mode);895 896//897// TTY utils898//899 900// Auto-detect if colors can be enabled based on terminal and environment901bool tty_can_use_colors();902 903//904// Model utils905//906 907struct common_sampler;908 909// note: defines the model, context, samplers, ets. lifetimes910struct common_init_result {911 common_init_result(common_params & params, bool model_only = false);912 ~common_init_result();913 914 llama_model * model();915 llama_context * context();916 917 common_sampler * sampler(llama_seq_id seq_id);918 void reset_samplers();919 920 std::vector<llama_adapter_lora_ptr> & lora();921 922private:923 struct impl;924 std::unique_ptr<impl> pimpl;925};926 927using common_init_result_ptr = std::unique_ptr<common_init_result>;928 929common_init_result_ptr common_init_from_params(common_params & params, bool model_only = false);930 931struct llama_model_params common_model_params_to_llama ( common_params & params);932struct llama_context_params common_context_params_to_llama(const common_params & params);933struct ggml_threadpool_params ggml_threadpool_params_from_cpu_params(const common_cpu_params & params);934 935// clear LoRA adapters from context, then apply new list of adapters936void common_set_adapter_lora(struct llama_context * ctx, std::vector<common_adapter_lora_info> & lora);937 938// model endpoint from env939std::string common_get_model_endpoint();940 941// for testing purposes942char * common_get_model_or_exit(int, char*[]);943 944//945// Context utils946//947 948enum common_context_seq_rm_type {949 COMMON_CONTEXT_SEQ_RM_TYPE_NO = 0, // seq_rm not supported (e.g. no memory module)950 COMMON_CONTEXT_SEQ_RM_TYPE_PART = 1, // can seq_rm partial sequences951 COMMON_CONTEXT_SEQ_RM_TYPE_FULL = 2, // can seq_rm full sequences only952 COMMON_CONTEXT_SEQ_RM_TYPE_RS = 3, // can seq_rm partial sequences, bounded by n_rs_seq953};954 955// check if the llama_context can remove sequences956// note: clears the memory of the context957common_context_seq_rm_type common_context_can_seq_rm(llama_context * ctx);958 959struct common_memory {960 llama_context * ctx_tgt = nullptr;961 llama_context * ctx_dft = nullptr;962 963 void init(llama_context * ctx_tgt, llama_context * ctx_dft = nullptr);964 965 // aborts execution on failure966 void seq_rm (llama_seq_id seq_id, llama_pos p0, llama_pos p1) const;967 void seq_add(llama_seq_id seq_id, llama_pos p0, llama_pos p1, llama_pos delta) const;968 void seq_cp (llama_seq_id seq_id_src, llama_seq_id seq_id_dst, llama_pos p0, llama_pos p1) const;969};970 971//972// Batch utils973//974 975void common_batch_clear(struct llama_batch & batch);976 977void common_batch_add(978 struct llama_batch & batch,979 llama_token id,980 llama_pos pos,981 const std::vector<llama_seq_id> & seq_ids,982 bool logits);983 984// decodes a single batch of tokens for a prompt and manages session tokens985//986// Note: We save state before the last token so that we can replay it to ensure987// compatibility with all memory types. Recurrent/hybrid models cannot remove988// tokens from memory, so this approach works across all model architectures.989bool common_prompt_batch_decode(990 struct llama_context * ctx,991 const std::vector<llama_token> & all_tokens,992 int n_new,993 int & n_past,994 int n_batch,995 std::string_view state_path,996 bool save_state);997 998// replays the last token after loading state to regenerate logits999// used after loading session state to ensure the sampling context has valid logits1000bool common_replay_last_token(struct llama_context * ctx, llama_token last_token, int32_t pos);1001 1002//1003// Vocab utils1004//1005 1006// tokenizes a string into a vector of tokens1007// should work similar to Python's `tokenizer.encode`1008std::vector<llama_token> common_tokenize(1009 const struct llama_context * ctx,1010 const std::string & text,1011 bool add_special,1012 bool parse_special = false);1013 1014std::vector<llama_token> common_tokenize(1015 const struct llama_vocab * vocab,1016 const std::string & text,1017 bool add_special,1018 bool parse_special = false);1019 1020// tokenizes a token into a piece, optionally renders special/control tokens1021// should work similar to Python's `tokenizer.id_to_piece`1022std::string common_token_to_piece(1023 const struct llama_context * ctx,1024 llama_token token,1025 bool special = true);1026 1027std::string common_token_to_piece(1028 const struct llama_vocab * vocab,1029 llama_token token,1030 bool special = true);1031 1032// detokenizes a vector of tokens into a string1033// should work similar to Python's `tokenizer.decode`1034// optionally renders special/control tokens1035std::string common_detokenize(1036 const struct llama_context * ctx,1037 const std::vector<llama_token> & tokens,1038 bool special = true);1039 1040std::string common_detokenize(1041 const struct llama_vocab * vocab,1042 const std::vector<llama_token> & tokens,1043 bool special = true);1044 1045//1046// Embedding utils1047//1048 1049// TODO: replace embd_norm with an enum1050void common_embd_normalize(const float * inp, float * out, int n, int embd_norm);1051 1052float common_embd_similarity_cos(const float * embd1, const float * embd2, int n);1053 1054//1055// Control vector utils1056//1057 1058struct common_control_vector_data {1059 int n_embd;1060 1061 // stores data for layers [1, n_layer] where n_layer = data.size() / n_embd1062 std::vector<float> data;1063};1064 1065struct common_control_vector_load_info {1066 float strength;1067 1068 std::string fname;1069};1070 1071// Load control vectors, scale each by strength, and add them together.1072// On error, returns {-1, empty}1073common_control_vector_data common_control_vector_load(const std::vector<common_control_vector_load_info> & load_infos);1074 1075//1076// Split utils1077//1078 1079namespace {1080 1081const char * const LLM_KV_SPLIT_NO = "split.no";1082const char * const LLM_KV_SPLIT_COUNT = "split.count";1083const char * const LLM_KV_SPLIT_TENSORS_COUNT = "split.tensors.count";1084 1085}1086 1087//1088// MoE utils1089//1090 1091const char * const LLM_FFN_EXPS_REGEX = "\\.ffn_(up|down|gate|gate_up)_(ch|)exps";1092 1093inline std::string llm_ffn_exps_block_regex(int idx) {1094 return string_format("blk\\.%d%s", idx, LLM_FFN_EXPS_REGEX);1095}1096 1097inline llama_model_tensor_buft_override llm_ffn_exps_cpu_override() {1098 return { LLM_FFN_EXPS_REGEX, ggml_backend_cpu_buffer_type() };1099}1100 1101//1102// training utils1103//1104 1105ggml_opt_dataset_t common_opt_dataset_init(struct llama_context * ctx, const std::vector<llama_token> & tokens, int64_t stride);1106 1107// "adamw" or "sgd" (case insensitive)1108enum ggml_opt_optimizer_type common_opt_get_optimizer(const char *);1109 1110//1111// prompt utils1112//1113 1114struct common_prompt_checkpoint {1115 int64_t n_tokens;1116 1117 // (optional) id of the task that created the checkpoint1118 int id_task = -1;1119 1120 llama_pos pos_min;1121 llama_pos pos_max;1122 1123 std::vector<uint8_t> data_tgt;1124 std::vector<uint8_t> data_dft;1125 1126 // (optional) speculative-decoding implementation state stashed with the checkpoint1127 // (e.g. eagle3's deferred-boundary g_embd row)1128 std::vector<uint8_t> data_spec;1129 1130 size_t size() const;1131 1132 bool empty() const;1133 void clear();1134 1135 void update_pos(1136 int64_t n_tokens,1137 llama_pos pos_min,1138 llama_pos pos_max);1139 1140 void update_tgt(1141 llama_context * ctx,1142 llama_seq_id seq_id,1143 llama_state_seq_flags flags);1144 1145 void update_dft(1146 llama_context * ctx,1147 llama_seq_id seq_id,1148 llama_state_seq_flags flags);1149 1150 void load_tgt(1151 llama_context * ctx,1152 llama_seq_id seq_id,1153 llama_state_seq_flags flags) const;1154 1155 void load_dft(1156 llama_context * ctx,1157 llama_seq_id seq_id,1158 llama_state_seq_flags flags) const;1159 1160 void clear_tgt();1161 void clear_dft();1162};1163 