Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#ifndef LLAMA_H2#define LLAMA_H3 4#include "ggml.h"5#include "ggml-cpu.h"6#include "ggml-backend.h"7#include "ggml-opt.h"8#include "gguf.h"9 10#include <stddef.h>11#include <stdint.h>12#include <stdio.h>13#include <stdbool.h>14 15#ifdef LLAMA_SHARED16# if defined(_WIN32) && !defined(__MINGW32__)17# ifdef LLAMA_BUILD18# define LLAMA_API __declspec(dllexport)19# else20# define LLAMA_API __declspec(dllimport)21# endif22# else23# define LLAMA_API __attribute__ ((visibility ("default")))24# endif25#else26# define LLAMA_API27#endif28 29#ifdef __GNUC__30# define DEPRECATED(func, hint) func __attribute__((deprecated(hint)))31#elif defined(_MSC_VER)32# define DEPRECATED(func, hint) __declspec(deprecated(hint)) func33#else34# define DEPRECATED(func, hint) func35#endif36 37#define LLAMA_DEFAULT_SEED 0xFFFFFFFF38 39#define LLAMA_TOKEN_NULL -140 41#define LLAMA_FILE_MAGIC_GGLA 0x67676c61u // 'ggla'42#define LLAMA_FILE_MAGIC_GGSN 0x6767736eu // 'ggsn'43#define LLAMA_FILE_MAGIC_GGSQ 0x67677371u // 'ggsq'44 45#define LLAMA_SESSION_MAGIC LLAMA_FILE_MAGIC_GGSN46#define LLAMA_SESSION_VERSION 947 48#define LLAMA_STATE_SEQ_MAGIC LLAMA_FILE_MAGIC_GGSQ49#define LLAMA_STATE_SEQ_VERSION 250 51#ifdef __cplusplus52extern "C" {53#endif54 55 //56 // C interface57 //58 // TODO: show sample usage59 //60 61 struct llama_vocab;62 struct llama_model;63 struct llama_context;64 struct llama_sampler;65 66 typedef struct llama_memory_i * llama_memory_t;67 68 typedef int32_t llama_pos;69 typedef int32_t llama_token;70 typedef int32_t llama_seq_id;71 72 enum llama_vocab_type {73 LLAMA_VOCAB_TYPE_NONE = 0, // For models without vocab74 LLAMA_VOCAB_TYPE_SPM = 1, // LLaMA tokenizer based on byte-level BPE with byte fallback75 LLAMA_VOCAB_TYPE_BPE = 2, // GPT-2 tokenizer based on byte-level BPE76 LLAMA_VOCAB_TYPE_WPM = 3, // BERT tokenizer based on WordPiece77 LLAMA_VOCAB_TYPE_UGM = 4, // T5 tokenizer based on Unigram78 LLAMA_VOCAB_TYPE_RWKV = 5, // RWKV tokenizer based on greedy tokenization79 LLAMA_VOCAB_TYPE_PLAMO2 = 6, // PLaMo-2 tokenizer based on Aho-Corasick with dynamic programming80 };81 82 enum llama_rope_type {83 LLAMA_ROPE_TYPE_NONE = -1,84 LLAMA_ROPE_TYPE_NORM = 0,85 LLAMA_ROPE_TYPE_NEOX = GGML_ROPE_TYPE_NEOX,86 LLAMA_ROPE_TYPE_MROPE = GGML_ROPE_TYPE_MROPE,87 LLAMA_ROPE_TYPE_IMROPE = GGML_ROPE_TYPE_IMROPE,88 LLAMA_ROPE_TYPE_VISION = GGML_ROPE_TYPE_VISION,89 };90 91 enum llama_token_type { //TODO: remove, required until per token attributes are available from GGUF file92 LLAMA_TOKEN_TYPE_UNDEFINED = 0,93 LLAMA_TOKEN_TYPE_NORMAL = 1,94 LLAMA_TOKEN_TYPE_UNKNOWN = 2,95 LLAMA_TOKEN_TYPE_CONTROL = 3,96 LLAMA_TOKEN_TYPE_USER_DEFINED = 4,97 LLAMA_TOKEN_TYPE_UNUSED = 5,98 LLAMA_TOKEN_TYPE_BYTE = 6,99 };100 101 enum llama_token_attr {102 LLAMA_TOKEN_ATTR_UNDEFINED = 0,103 LLAMA_TOKEN_ATTR_UNKNOWN = 1 << 0,104 LLAMA_TOKEN_ATTR_UNUSED = 1 << 1,105 LLAMA_TOKEN_ATTR_NORMAL = 1 << 2,106 LLAMA_TOKEN_ATTR_CONTROL = 1 << 3, // SPECIAL?107 LLAMA_TOKEN_ATTR_USER_DEFINED = 1 << 4,108 LLAMA_TOKEN_ATTR_BYTE = 1 << 5,109 LLAMA_TOKEN_ATTR_NORMALIZED = 1 << 6,110 LLAMA_TOKEN_ATTR_LSTRIP = 1 << 7,111 LLAMA_TOKEN_ATTR_RSTRIP = 1 << 8,112 LLAMA_TOKEN_ATTR_SINGLE_WORD = 1 << 9,113 };114 115 // model file types116 enum llama_ftype {117 LLAMA_FTYPE_ALL_F32 = 0,118 LLAMA_FTYPE_MOSTLY_F16 = 1, // except 1d tensors119 LLAMA_FTYPE_MOSTLY_Q4_0 = 2, // except 1d tensors120 LLAMA_FTYPE_MOSTLY_Q4_1 = 3, // except 1d tensors121 // LLAMA_FTYPE_MOSTLY_Q4_1_SOME_F16 = 4, // tok_embeddings.weight and output.weight are F16122 // LLAMA_FTYPE_MOSTLY_Q4_2 = 5, // support has been removed123 // LLAMA_FTYPE_MOSTLY_Q4_3 = 6, // support has been removed124 LLAMA_FTYPE_MOSTLY_Q8_0 = 7, // except 1d tensors125 LLAMA_FTYPE_MOSTLY_Q5_0 = 8, // except 1d tensors126 LLAMA_FTYPE_MOSTLY_Q5_1 = 9, // except 1d tensors127 LLAMA_FTYPE_MOSTLY_Q2_K = 10, // except 1d tensors128 LLAMA_FTYPE_MOSTLY_Q3_K_S = 11, // except 1d tensors129 LLAMA_FTYPE_MOSTLY_Q3_K_M = 12, // except 1d tensors130 LLAMA_FTYPE_MOSTLY_Q3_K_L = 13, // except 1d tensors131 LLAMA_FTYPE_MOSTLY_Q4_K_S = 14, // except 1d tensors132 LLAMA_FTYPE_MOSTLY_Q4_K_M = 15, // except 1d tensors133 LLAMA_FTYPE_MOSTLY_Q5_K_S = 16, // except 1d tensors134 LLAMA_FTYPE_MOSTLY_Q5_K_M = 17, // except 1d tensors135 LLAMA_FTYPE_MOSTLY_Q6_K = 18, // except 1d tensors136 LLAMA_FTYPE_MOSTLY_IQ2_XXS = 19, // except 1d tensors137 LLAMA_FTYPE_MOSTLY_IQ2_XS = 20, // except 1d tensors138 LLAMA_FTYPE_MOSTLY_Q2_K_S = 21, // except 1d tensors139 LLAMA_FTYPE_MOSTLY_IQ3_XS = 22, // except 1d tensors140 LLAMA_FTYPE_MOSTLY_IQ3_XXS = 23, // except 1d tensors141 LLAMA_FTYPE_MOSTLY_IQ1_S = 24, // except 1d tensors142 LLAMA_FTYPE_MOSTLY_IQ4_NL = 25, // except 1d tensors143 LLAMA_FTYPE_MOSTLY_IQ3_S = 26, // except 1d tensors144 LLAMA_FTYPE_MOSTLY_IQ3_M = 27, // except 1d tensors145 LLAMA_FTYPE_MOSTLY_IQ2_S = 28, // except 1d tensors146 LLAMA_FTYPE_MOSTLY_IQ2_M = 29, // except 1d tensors147 LLAMA_FTYPE_MOSTLY_IQ4_XS = 30, // except 1d tensors148 LLAMA_FTYPE_MOSTLY_IQ1_M = 31, // except 1d tensors149 LLAMA_FTYPE_MOSTLY_BF16 = 32, // except 1d tensors150 //LLAMA_FTYPE_MOSTLY_Q4_0_4_4 = 33, // removed from gguf files, use Q4_0 and runtime repack151 //LLAMA_FTYPE_MOSTLY_Q4_0_4_8 = 34, // removed from gguf files, use Q4_0 and runtime repack152 //LLAMA_FTYPE_MOSTLY_Q4_0_8_8 = 35, // removed from gguf files, use Q4_0 and runtime repack153 LLAMA_FTYPE_MOSTLY_TQ1_0 = 36, // except 1d tensors154 LLAMA_FTYPE_MOSTLY_TQ2_0 = 37, // except 1d tensors155 LLAMA_FTYPE_MOSTLY_MXFP4_MOE = 38, // except 1d tensors156 LLAMA_FTYPE_MOSTLY_NVFP4 = 39, // except 1d tensors157 LLAMA_FTYPE_MOSTLY_Q1_0 = 40, // except 1d tensors158 LLAMA_FTYPE_MOSTLY_Q2_0 = 41, // except 1d tensors159 160 LLAMA_FTYPE_GUESSED = 1024, // not specified in the model file161 };162 163 // Get the model file type (quantization) as a string, e.g. "Q8_0" or "Q4_K - Medium"164 LLAMA_API const char * llama_ftype_name(enum llama_ftype ftype);165 166 enum llama_rope_scaling_type {167 LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED = -1,168 LLAMA_ROPE_SCALING_TYPE_NONE = 0,169 LLAMA_ROPE_SCALING_TYPE_LINEAR = 1,170 LLAMA_ROPE_SCALING_TYPE_YARN = 2,171 LLAMA_ROPE_SCALING_TYPE_LONGROPE = 3,172 LLAMA_ROPE_SCALING_TYPE_MAX_VALUE = LLAMA_ROPE_SCALING_TYPE_LONGROPE,173 };174 175 enum llama_pooling_type {176 LLAMA_POOLING_TYPE_UNSPECIFIED = -1,177 LLAMA_POOLING_TYPE_NONE = 0,178 LLAMA_POOLING_TYPE_MEAN = 1,179 LLAMA_POOLING_TYPE_CLS = 2,180 LLAMA_POOLING_TYPE_LAST = 3,181 LLAMA_POOLING_TYPE_RANK = 4, // used by reranking models to attach the classification head to the graph182 };183 184 enum llama_attention_type {185 LLAMA_ATTENTION_TYPE_UNSPECIFIED = -1,186 LLAMA_ATTENTION_TYPE_CAUSAL = 0,187 LLAMA_ATTENTION_TYPE_NON_CAUSAL = 1,188 };189 190 enum llama_flash_attn_type {191 LLAMA_FLASH_ATTN_TYPE_AUTO = -1,192 LLAMA_FLASH_ATTN_TYPE_DISABLED = 0,193 LLAMA_FLASH_ATTN_TYPE_ENABLED = 1,194 };195 196 LLAMA_API const char * llama_flash_attn_type_name(enum llama_flash_attn_type flash_attn_type);197 198 enum llama_split_mode {199 LLAMA_SPLIT_MODE_NONE = 0, // single GPU200 LLAMA_SPLIT_MODE_LAYER = 1, // split layers and KV across GPUs201 LLAMA_SPLIT_MODE_ROW = 2, // split layers and KV across GPUs, use tensor parallelism if supported202 LLAMA_SPLIT_MODE_TENSOR = 3,203 };204 205 enum llama_load_mode {206 LLAMA_LOAD_MODE_NONE = 0, // no special loading mode207 LLAMA_LOAD_MODE_MMAP = 1, // memory map the model208 LLAMA_LOAD_MODE_MLOCK = 2, // force system to keep model in RAM rather than swapping or compressing209 LLAMA_LOAD_MODE_MMAP_MLOCK = 3, // mmap + force system to keep model in RAM rather than swapping or compressing210 LLAMA_LOAD_MODE_DIRECT_IO = 4, // use direct I/O if available211 };212 213 LLAMA_API const char * llama_load_mode_name(enum llama_load_mode load_mode);214 LLAMA_API enum llama_load_mode llama_load_mode_from_str(const char * str);215 216 enum llama_context_type {217 LLAMA_CONTEXT_TYPE_DEFAULT = 0,218 LLAMA_CONTEXT_TYPE_MTP = 1,219 };220 221 // TODO: simplify (https://github.com/ggml-org/llama.cpp/pull/9294#pullrequestreview-2286561979)222 typedef struct llama_token_data {223 llama_token id; // token id224 float logit; // log-odds of the token225 float p; // probability of the token226 } llama_token_data;227 228 typedef struct llama_token_data_array {229 // TODO: consider SoA230 // NOTE: this pointer can be modified by the samplers231 llama_token_data * data;232 size_t size;233 int64_t selected; // this is the index in the data array (i.e. not the token id)234 bool sorted; // note: do not assume the data is sorted - always check this flag235 } llama_token_data_array;236 237 typedef bool (*llama_progress_callback)(float progress, void * user_data);238 239 // Input data for llama_encode/llama_decode240 // A llama_batch object can contain input about one or many sequences241 // The provided arrays (i.e. token, embd, pos, etc.) must have size of n_tokens242 //243 // - token : the token ids of the input (used when embd is NULL)244 // - embd : token embeddings (i.e. float vector of size n_embd) (used when token is NULL)245 // - pos : the positions of the respective token in the sequence246 // (if set to NULL, the token position will be tracked automatically by llama_encode/llama_decode)247 // - seq_id : the sequence to which the respective token belongs248 // (if set to NULL, the sequence ID will be assumed to be 0)249 // - logits : if zero, the logits (and/or the embeddings) for the respective token will not be output250 // (if set to NULL:251 // - if embeddings: all tokens are output252 // - if not: only the last token is output253 // )254 //255 typedef struct llama_batch {256 int32_t n_tokens;257 258 llama_token * token;259 float * embd;260 llama_pos * pos;261 int32_t * n_seq_id;262 llama_seq_id ** seq_id;263 int8_t * logits; // TODO: rename this to "output"264 } llama_batch;265 266 enum llama_model_kv_override_type {267 LLAMA_KV_OVERRIDE_TYPE_INT,268 LLAMA_KV_OVERRIDE_TYPE_FLOAT,269 LLAMA_KV_OVERRIDE_TYPE_BOOL,270 LLAMA_KV_OVERRIDE_TYPE_STR,271 };272 273 enum llama_model_meta_key {274 LLAMA_MODEL_META_KEY_SAMPLING_SEQUENCE,275 LLAMA_MODEL_META_KEY_SAMPLING_TOP_K,276 LLAMA_MODEL_META_KEY_SAMPLING_TOP_P,277 LLAMA_MODEL_META_KEY_SAMPLING_MIN_P,278 LLAMA_MODEL_META_KEY_SAMPLING_XTC_PROBABILITY,279 LLAMA_MODEL_META_KEY_SAMPLING_XTC_THRESHOLD,280 LLAMA_MODEL_META_KEY_SAMPLING_TEMP,281 LLAMA_MODEL_META_KEY_SAMPLING_PENALTY_LAST_N,282 LLAMA_MODEL_META_KEY_SAMPLING_PENALTY_REPEAT,283 LLAMA_MODEL_META_KEY_SAMPLING_MIROSTAT,284 LLAMA_MODEL_META_KEY_SAMPLING_MIROSTAT_TAU,285 LLAMA_MODEL_META_KEY_SAMPLING_MIROSTAT_ETA,286 };287 288 struct llama_model_kv_override {289 enum llama_model_kv_override_type tag;290 291 char key[128];292 293 union {294 int64_t val_i64;295 double val_f64;296 bool val_bool;297 char val_str[128];298 };299 };300 301 struct llama_model_tensor_buft_override {302 const char * pattern;303 ggml_backend_buffer_type_t buft;304 };305 306 struct llama_model_params {307 // NULL-terminated list of devices to use for offloading (if NULL, all available devices are used)308 ggml_backend_dev_t * devices;309 310 // NULL-terminated list of buffer types to use for tensors that match a pattern311 const struct llama_model_tensor_buft_override * tensor_buft_overrides;312 313 int32_t n_gpu_layers; // number of layers to store in VRAM, a negative value means all layers314 enum llama_split_mode split_mode; // how to split the model across multiple GPUs315 enum llama_load_mode load_mode; // how to load the model316 317 // the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE318 int32_t main_gpu;319 320 // proportion of the model (layers or rows) to offload to each GPU, size: llama_max_devices()321 const float * tensor_split;322 323 // Called with a progress value between 0.0 and 1.0. Pass NULL to disable.324 // If the provided progress_callback returns true, model loading continues.325 // If it returns false, model loading is immediately aborted.326 llama_progress_callback progress_callback;327 328 // context pointer passed to the progress callback329 void * progress_callback_user_data;330 331 // override key-value pairs of the model meta data332 const struct llama_model_kv_override * kv_overrides;333 334 // Keep the booleans together to avoid misalignment during copy-by-value.335 bool vocab_only; // only load the vocabulary, no weights336 bool check_tensors; // validate model tensor data337 bool use_extra_bufts; // use extra buffer types (used for weight repacking)338 bool no_host; // bypass host buffer allowing extra buffers to be used339 bool no_alloc; // only load metadata and simulate memory allocations340 bool load_mtp; // whether to load MTP layers341 };342 343 struct llama_sampler_seq_config {344 llama_seq_id seq_id;345 struct llama_sampler * sampler;346 };347 348 // NOTE: changing the default values of parameters marked as [EXPERIMENTAL] may cause crashes or incorrect results in certain configurations349 // https://github.com/ggml-org/llama.cpp/pull/7544350 struct llama_context_params {351 uint32_t n_ctx; // text context, 0 = from model352 uint32_t n_batch; // logical maximum batch size that can be submitted to llama_decode353 uint32_t n_ubatch; // physical maximum batch size354 uint32_t n_seq_max; // max number of sequences (i.e. distinct states for recurrent models)355 uint32_t n_rs_seq; // number of recurrent-state snapshots per seq for rollback (0 = no rollback) [EXPERIMENTAL]356 uint32_t n_outputs_max; // max outputs in a ubatch (0 = n_batch)357 uint32_t n_outputs_max_per_seq; // max outputs per sequence (0 = n_outputs_max)358 int32_t n_threads; // number of threads to use for generation359 int32_t n_threads_batch; // number of threads to use for batch processing360 361 enum llama_context_type ctx_type; // set the context type (e.g. MTP)362 enum llama_rope_scaling_type rope_scaling_type; // RoPE scaling type, from `enum llama_rope_scaling_type`363 enum llama_pooling_type pooling_type; // whether to pool (sum) embedding results by sequence id364 enum llama_attention_type attention_type; // attention type to use for embeddings365 enum llama_flash_attn_type flash_attn_type; // when to enable Flash Attention366 367 // ref: https://github.com/ggml-org/llama.cpp/pull/2054368 float rope_freq_base; // RoPE base frequency, 0 = from model369 float rope_freq_scale; // RoPE frequency scaling factor, 0 = from model370 float yarn_ext_factor; // YaRN extrapolation mix factor, negative = from model371 float yarn_attn_factor; // YaRN magnitude scaling factor372 float yarn_beta_fast; // YaRN low correction dim373 float yarn_beta_slow; // YaRN high correction dim374 uint32_t yarn_orig_ctx; // YaRN original context size375 float defrag_thold; // [DEPRECATED] defragment the KV cache if holes/size > thold, <= 0 disabled (default)376 377 ggml_backend_sched_eval_callback cb_eval;378 void * cb_eval_user_data;379 380 enum ggml_type type_k; // data type for K cache [EXPERIMENTAL]381 enum ggml_type type_v; // data type for V cache [EXPERIMENTAL]382 383 // Abort callback384 // if it returns true, execution of llama_decode() will be aborted385 // currently works only with CPU execution386 ggml_abort_callback abort_callback;387 void * abort_callback_data;388 389 // Keep the booleans together and at the end of the struct to avoid misalignment during copy-by-value.390 bool embeddings; // if true, extract embeddings (together with logits)391 bool offload_kqv; // offload the KQV ops (including the KV cache) to GPU392 bool no_perf; // measure performance timings393 bool op_offload; // offload host tensor operations to device394 bool swa_full; // use full-size SWA cache (https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)395 // NOTE: setting to false when n_seq_max > 1 can cause bad performance in some cases396 // ref: https://github.com/ggml-org/llama.cpp/pull/13845#issuecomment-2924800573397 bool kv_unified; // use a unified buffer across the input sequences when computing the attention398 // try to disable when n_seq_max > 1 for improved performance when the sequences do not share a large prefix399 // ref: https://github.com/ggml-org/llama.cpp/pull/14363400 401 // [EXPERIMENTAL]402 // backend sampler chain configuration (make sure the caller keeps the sampler chains alive)403 // note: the samplers must be sampler chains (i.e. use llama_sampler_chain_init)404 struct llama_sampler_seq_config * samplers;405 size_t n_samplers;406 407 // a source/target/parent context408 // can be utilized in various ways, for example by sharing results or llama_memory between 2 contexts409 struct llama_context * ctx_other;410 };411 412 struct llama_model_tensor_override {413 const char * pattern;414 enum ggml_type type;415 };416 417 struct llama_model_imatrix_data {418 const char * name;419 const float * data;420 size_t size;421 };422 423 // model quantization parameters424 typedef struct llama_model_quantize_params {425 int32_t nthread; // number of threads to use for quantizing, if <=0 will use std::thread::hardware_concurrency()426 enum llama_ftype ftype; // quantize to this llama_ftype427 enum ggml_type output_tensor_type; // output tensor type428 enum ggml_type token_embedding_type; // token embeddings tensor type429 bool allow_requantize; // allow quantizing non-f32/f16 tensors430 bool quantize_output_tensor; // quantize output.weight431 bool only_copy; // only copy tensors - ftype, allow_requantize and quantize_output_tensor are ignored432 bool pure; // quantize all tensors to the default type433 bool keep_split; // quantize to the same number of shards434 bool dry_run; // calculate and show the final quantization size without performing quantization435 const struct llama_model_imatrix_data * imatrix; // pointer to importance matrix data436 const struct llama_model_kv_override * kv_overrides; // pointer to kv overrides437 const struct llama_model_tensor_override * tt_overrides; // pointer to tensor overrides438 const int32_t * prune_layers; // pointer to layer indices to prune439 } llama_model_quantize_params;440 441 typedef struct llama_logit_bias {442 llama_token token;443 float bias;444 } llama_logit_bias;445 446 typedef struct llama_sampler_chain_params {447 bool no_perf; // whether to measure performance timings448 } llama_sampler_chain_params;449 450 // used in chat template451 typedef struct llama_chat_message {452 const char * role;453 const char * content;454 } llama_chat_message;455 456 // lora adapter457 struct llama_adapter_lora;458 459 // Helpers for getting default parameters460 // TODO: update API to start accepting pointers to params structs (https://github.com/ggml-org/llama.cpp/discussions/9172)461 LLAMA_API struct llama_model_params llama_model_default_params(void);462 LLAMA_API struct llama_context_params llama_context_default_params(void);463 LLAMA_API struct llama_sampler_chain_params llama_sampler_chain_default_params(void);464 LLAMA_API struct llama_model_quantize_params llama_model_quantize_default_params(void);465 466 // Initialize the llama + ggml backend467 // If numa is true, use NUMA optimizations468 // Call once at the start of the program469 LLAMA_API void llama_backend_init(void);470 471 // Call once at the end of the program - currently only used for MPI472 LLAMA_API void llama_backend_free(void);473 474 //optional:475 LLAMA_API void llama_numa_init(enum ggml_numa_strategy numa);476 477 // Optional: an auto threadpool gets created in ggml if not passed explicitly478 LLAMA_API void llama_attach_threadpool(479 struct llama_context * ctx,480 ggml_threadpool_t threadpool,481 ggml_threadpool_t threadpool_batch);482 483 LLAMA_API void llama_detach_threadpool(struct llama_context * ctx);484 485 typedef void (*llama_model_set_tensor_data_t)(struct ggml_tensor * tensor, void * userdata);486 487 // Create a new model from GGUF metadata as well as a function to set the tensor data488 // - tensors are created as GGML_TYPE_F32 by default,489 // override by adding a tensor with the same name but a different name to the context490 LLAMA_API struct llama_model * llama_model_init_from_user(491 struct gguf_context * metadata,492 llama_model_set_tensor_data_t set_tensor_data, // function to initialize tensor data with493 void * set_tensor_data_ud, // userdata for function494 struct llama_model_params params);495 496 DEPRECATED(LLAMA_API struct llama_model * llama_load_model_from_file(497 const char * path_model,498 struct llama_model_params params),499 "use llama_model_load_from_file instead");500 501 // Load a model from a file502 // If the file is split into multiple parts, the file name must follow this pattern: <name>-%05d-of-%05d.gguf503 // If the split file name does not follow this pattern, use llama_model_load_from_splits504 LLAMA_API struct llama_model * llama_model_load_from_file(505 const char * path_model,506 struct llama_model_params params);507 508 // Load a model from an open FILE pointer509 LLAMA_API struct llama_model * llama_model_load_from_file_ptr(510 FILE * file,511 struct llama_model_params params);512 513 // Load a model from multiple splits (support custom naming scheme)514 // The paths must be in the correct order515 LLAMA_API struct llama_model * llama_model_load_from_splits(516 const char ** paths,517 size_t n_paths,518 struct llama_model_params params);519 520 LLAMA_API void llama_model_save_to_file(521 const struct llama_model * model,522 const char * path_model);523 524 DEPRECATED(LLAMA_API void llama_free_model(struct llama_model * model),525 "use llama_model_free instead");526 527 LLAMA_API void llama_model_free(struct llama_model * model);528 529 LLAMA_API struct llama_context * llama_init_from_model(530 struct llama_model * model,531 struct llama_context_params params);532 533 DEPRECATED(LLAMA_API struct llama_context * llama_new_context_with_model(534 struct llama_model * model,535 struct llama_context_params params),536 "use llama_init_from_model instead");537 538 // Frees all allocated memory539 LLAMA_API void llama_free(struct llama_context * ctx);540 541 LLAMA_API int64_t llama_time_us(void);542 543 LLAMA_API size_t llama_max_devices(void);544 LLAMA_API size_t llama_max_parallel_sequences(void);545 LLAMA_API size_t llama_max_tensor_buft_overrides(void);546 547 LLAMA_API bool llama_supports_mmap (void);548 LLAMA_API bool llama_supports_mlock (void);549 LLAMA_API bool llama_supports_gpu_offload(void);550 LLAMA_API bool llama_supports_rpc (void);551 552 // NOTE: After creating a llama_context, it is recommended to query the actual values using these functions553 // In some cases the requested values via llama_context_params may differ from the actual values used by the context554 // ref: https://github.com/ggml-org/llama.cpp/pull/17046#discussion_r2503085732555 LLAMA_API uint32_t llama_n_ctx (const struct llama_context * ctx);556 LLAMA_API uint32_t llama_n_ctx_seq (const struct llama_context * ctx);557 LLAMA_API uint32_t llama_n_batch (const struct llama_context * ctx);558 LLAMA_API uint32_t llama_n_ubatch (const struct llama_context * ctx);559 LLAMA_API uint32_t llama_n_seq_max (const struct llama_context * ctx);560 LLAMA_API uint32_t llama_n_rs_seq (const struct llama_context * ctx);561 562 DEPRECATED(LLAMA_API int32_t llama_n_ctx_train(const struct llama_model * model), "use llama_model_n_ctx_train instead");563 DEPRECATED(LLAMA_API int32_t llama_n_embd (const struct llama_model * model), "use llama_model_n_embd instead");564 DEPRECATED(LLAMA_API int32_t llama_n_layer (const struct llama_model * model), "use llama_model_n_layer instead");565 DEPRECATED(LLAMA_API int32_t llama_n_head (const struct llama_model * model), "use llama_model_n_head instead");566 567 DEPRECATED(LLAMA_API int32_t llama_n_vocab (const struct llama_vocab * vocab), "use llama_vocab_n_tokens instead");568 569 LLAMA_API const struct llama_model * llama_get_model (const struct llama_context * ctx);570 LLAMA_API llama_memory_t llama_get_memory (const struct llama_context * ctx);571 LLAMA_API enum llama_pooling_type llama_pooling_type(const struct llama_context * ctx); // TODO: rename to llama_get_pooling_type572 573 LLAMA_API const struct llama_vocab * llama_model_get_vocab(const struct llama_model * model);574 LLAMA_API enum llama_rope_type llama_model_rope_type(const struct llama_model * model);575 576 LLAMA_API int32_t llama_model_n_ctx_train (const struct llama_model * model);577 LLAMA_API int32_t llama_model_n_embd (const struct llama_model * model);578 LLAMA_API int32_t llama_model_n_embd_inp (const struct llama_model * model);579 LLAMA_API int32_t llama_model_n_embd_out (const struct llama_model * model);580 LLAMA_API int32_t llama_model_n_layer (const struct llama_model * model);581 LLAMA_API int32_t llama_model_n_layer_nextn(const struct llama_model * model);582 LLAMA_API int32_t llama_model_n_head (const struct llama_model * model);583 LLAMA_API int32_t llama_model_n_head_kv (const struct llama_model * model);584 LLAMA_API int32_t llama_model_n_swa (const struct llama_model * model);585 586 // Get the model's RoPE frequency scaling factor587 LLAMA_API float llama_model_rope_freq_scale_train(const struct llama_model * model);588 589 // Returns the number of classifier outputs (only valid for classifier models)590 // Undefined behavior for non-classifier models591 LLAMA_API uint32_t llama_model_n_cls_out(const struct llama_model * model);592 593 // Returns label of classifier output by index (<n_cls_out). Returns nullptr if no label provided594 LLAMA_API const char * llama_model_cls_label(const struct llama_model * model, uint32_t i);595 596 LLAMA_API enum llama_vocab_type llama_vocab_type(const struct llama_vocab * vocab);597 598 LLAMA_API int32_t llama_vocab_n_tokens(const struct llama_vocab * vocab);599 600 // Functions to access the model's GGUF metadata scalar values601 // - The functions return the length of the string on success, or -1 on failure602 // - The output string is always null-terminated and cleared on failure603 // - When retrieving a string, an extra byte must be allocated to account for the null terminator604 // - GGUF array values are not supported by these functions605 606 // Get metadata value as a string by key name607 LLAMA_API int32_t llama_model_meta_val_str(const struct llama_model * model, const char * key, char * buf, size_t buf_size);608 609 // Get the number of metadata key/value pairs610 LLAMA_API int32_t llama_model_meta_count(const struct llama_model * model);611 612 // Get sampling metadata key name. Returns nullptr if the key is invalid613 LLAMA_API const char * llama_model_meta_key_str(enum llama_model_meta_key key);614 615 // Get metadata key name by index616 LLAMA_API int32_t llama_model_meta_key_by_index(const struct llama_model * model, int32_t i, char * buf, size_t buf_size);617 618 // Get metadata value as a string by index619 LLAMA_API int32_t llama_model_meta_val_str_by_index(const struct llama_model * model, int32_t i, char * buf, size_t buf_size);620 621 // Get a string describing the model type622 LLAMA_API int32_t llama_model_desc(const struct llama_model * model, char * buf, size_t buf_size);623 624 // Get the model file type (quantization), e.g. LLAMA_FTYPE_MOSTLY_Q8_0625 LLAMA_API enum llama_ftype llama_model_ftype(const struct llama_model * model);626 627 // Returns the total size of all the tensors in the model in bytes628 LLAMA_API uint64_t llama_model_size(const struct llama_model * model);629 630 // Get the default chat template. Returns nullptr if not available631 // If name is NULL, returns the default chat template632 LLAMA_API const char * llama_model_chat_template(const struct llama_model * model, const char * name);633 634 // Returns the total number of parameters in the model635 LLAMA_API uint64_t llama_model_n_params(const struct llama_model * model);636 637 // Returns true if the model contains an encoder that requires llama_encode() call638 LLAMA_API bool llama_model_has_encoder(const struct llama_model * model);639 640 // Returns true if the model contains a decoder that requires llama_decode() call641 LLAMA_API bool llama_model_has_decoder(const struct llama_model * model);642 643 // For encoder-decoder models, this function returns id of the token that must be provided644 // to the decoder to start generating output sequence. For other models, it returns -1.645 LLAMA_API llama_token llama_model_decoder_start_token(const struct llama_model * model);646 647 // Returns true if the model is recurrent (like Mamba, RWKV, etc.)648 LLAMA_API bool llama_model_is_recurrent(const struct llama_model * model);649 650 // Returns true if the model is hybrid (like Jamba, Granite, etc.)651 LLAMA_API bool llama_model_is_hybrid(const struct llama_model * model);652 653 // Returns true if the model is diffusion-based (like LLaDA, Dream, etc.)654 LLAMA_API bool llama_model_is_diffusion(const struct llama_model * model);655 656 // Returns 0 on success657 LLAMA_API uint32_t llama_model_quantize(658 const char * fname_inp,659 const char * fname_out,660 const llama_model_quantize_params * params);661 662 //663 // Adapters664 //665 666 // Load a LoRA adapter from file667 // The adapter is valid as long as the associated model is not freed668 LLAMA_API struct llama_adapter_lora * llama_adapter_lora_init(669 struct llama_model * model,670 const char * path_lora);671 672 // Functions to access the adapter's GGUF metadata scalar values673 // - The functions return the length of the string on success, or -1 on failure674 // - The output string is always null-terminated and cleared on failure675 // - When retrieving a string, an extra byte must be allocated to account for the null terminator676 // - GGUF array values are not supported by these functions677 678 // Get metadata value as a string by key name679 LLAMA_API int32_t llama_adapter_meta_val_str(const struct llama_adapter_lora * adapter, const char * key, char * buf, size_t buf_size);680 681 // Get the number of metadata key/value pairs682 LLAMA_API int32_t llama_adapter_meta_count(const struct llama_adapter_lora * adapter);683 684 // Get metadata key name by index685 LLAMA_API int32_t llama_adapter_meta_key_by_index(const struct llama_adapter_lora * adapter, int32_t i, char * buf, size_t buf_size);686 687 // Get metadata value as a string by index688 LLAMA_API int32_t llama_adapter_meta_val_str_by_index(const struct llama_adapter_lora * adapter, int32_t i, char * buf, size_t buf_size);689 690 // Manually free a LoRA adapter691 // NOTE: loaded adapters that are not manually freed will be freed when the associated model is deleted692 LLAMA_API void llama_adapter_lora_free(struct llama_adapter_lora * adapter);693 694 // Get the invocation tokens if the current lora is an alora695 LLAMA_API uint64_t llama_adapter_get_alora_n_invocation_tokens(const struct llama_adapter_lora * adapter);696 LLAMA_API const llama_token * llama_adapter_get_alora_invocation_tokens (const struct llama_adapter_lora * adapter);697 698 // The following functions operate on a llama_context, hence the naming: llama_verb_...699 700 // Set LoRa adapters on the context. Will only modify if the adapters currently in context are different.701 LLAMA_API int32_t llama_set_adapters_lora(702 struct llama_context * ctx,703 struct llama_adapter_lora ** adapters,704 size_t n_adapters,705 float * scales);706 707 // Apply a loaded control vector to a llama_context, or if data is NULL, clear708 // the currently loaded vector.709 // n_embd should be the size of a single layer's control, and data should point710 // to an n_embd x n_layers buffer starting from layer 1.711 // il_start and il_end are the layer range the vector should apply to (both inclusive)712 // See llama_control_vector_load in common to load a control vector.713 LLAMA_API int32_t llama_set_adapter_cvec(714 struct llama_context * ctx,715 const float * data,716 size_t len,717 int32_t n_embd,718 int32_t il_start,719 int32_t il_end);720 721 //722 // Memory723 //724 725 // Clear the memory contents726 // If data == true, the data buffers will also be cleared together with the metadata727 LLAMA_API void llama_memory_clear(728 llama_memory_t mem,729 bool data);730 731 // Removes all tokens that belong to the specified sequence and have positions in [p0, p1)732 // Returns false if a partial sequence cannot be removed. Removing a whole sequence never fails733 // seq_id < 0 : match any sequence734 // p0 < 0 : [0, p1]735 // p1 < 0 : [p0, inf)736 LLAMA_API bool llama_memory_seq_rm(737 llama_memory_t mem,738 llama_seq_id seq_id,739 llama_pos p0,740 llama_pos p1);741 742 // Copy all tokens that belong to the specified sequence to another sequence743 // p0 < 0 : [0, p1]744 // p1 < 0 : [p0, inf)745 LLAMA_API void llama_memory_seq_cp(746 llama_memory_t mem,747 llama_seq_id seq_id_src,748 llama_seq_id seq_id_dst,749 llama_pos p0,750 llama_pos p1);751 752 // Removes all tokens that do not belong to the specified sequence753 LLAMA_API void llama_memory_seq_keep(754 llama_memory_t mem,755 llama_seq_id seq_id);756 757 // Adds relative position "delta" to all tokens that belong to the specified sequence and have positions in [p0, p1)758 // p0 < 0 : [0, p1]759 // p1 < 0 : [p0, inf)760 LLAMA_API void llama_memory_seq_add(761 llama_memory_t mem,762 llama_seq_id seq_id,763 llama_pos p0,764 llama_pos p1,765 llama_pos delta);766 767 // Integer division of the positions by factor of `d > 1`768 // p0 < 0 : [0, p1]769 // p1 < 0 : [p0, inf)770 LLAMA_API void llama_memory_seq_div(771 llama_memory_t mem,772 llama_seq_id seq_id,773 llama_pos p0,774 llama_pos p1,775 int d);776 777 // Returns the smallest position present in the memory for the specified sequence778 // This is typically non-zero only for SWA caches779 // Note that all positions in the range [pos_min, pos_max] are guaranteed to be present in the memory780 // Return -1 if the sequence is empty781 LLAMA_API llama_pos llama_memory_seq_pos_min(782 llama_memory_t mem,783 llama_seq_id seq_id);784 785 // Returns the largest position present in the memory for the specified sequence786 // Note that all positions in the range [pos_min, pos_max] are guaranteed to be present in the memory787 // Return -1 if the sequence is empty788 LLAMA_API llama_pos llama_memory_seq_pos_max(789 llama_memory_t mem,790 llama_seq_id seq_id);791 792 // Check if the memory supports shifting793 LLAMA_API bool llama_memory_can_shift(llama_memory_t mem);794 795 //796 // State / sessions797 //798 799 // Returns the *actual* size in bytes of the state800 // (logits, embedding and memory)801 // Only use when saving the state, not when restoring it, otherwise the size may be too small.802 LLAMA_API size_t llama_state_get_size(struct llama_context * ctx);803 LLAMA_API DEPRECATED(size_t llama_get_state_size(struct llama_context * ctx),804 "use llama_state_get_size instead");805 806 // Copies the state to the specified destination address.807 // Destination needs to have allocated enough memory.808 // Returns the number of bytes copied809 LLAMA_API size_t llama_state_get_data(810 struct llama_context * ctx,811 uint8_t * dst,812 size_t size);813 LLAMA_API DEPRECATED(size_t llama_copy_state_data(814 struct llama_context * ctx,815 uint8_t * dst),816 "use llama_state_get_data instead");817 818 // Set the state reading from the specified address819 // Returns the number of bytes read820 LLAMA_API size_t llama_state_set_data(821 struct llama_context * ctx,822 const uint8_t * src,823 size_t size);824 LLAMA_API DEPRECATED(size_t llama_set_state_data(825 struct llama_context * ctx,826 const uint8_t * src),827 "use llama_state_set_data instead");828 829 // Save/load session file830 LLAMA_API bool llama_state_load_file(831 struct llama_context * ctx,832 const char * path_session,833 llama_token * tokens_out,834 size_t n_token_capacity,835 size_t * n_token_count_out);836 LLAMA_API DEPRECATED(bool llama_load_session_file(837 struct llama_context * ctx,838 const char * path_session,839 llama_token * tokens_out,840 size_t n_token_capacity,841 size_t * n_token_count_out),842 "use llama_state_load_file instead");843 844 LLAMA_API bool llama_state_save_file(845 struct llama_context * ctx,846 const char * path_session,847 const llama_token * tokens,848 size_t n_token_count);849 LLAMA_API DEPRECATED(bool llama_save_session_file(850 struct llama_context * ctx,851 const char * path_session,852 const llama_token * tokens,853 size_t n_token_count),854 "use llama_state_save_file instead");855 856 // Get the exact size needed to copy the state of a single sequence857 LLAMA_API size_t llama_state_seq_get_size(858 struct llama_context * ctx,859 llama_seq_id seq_id);860 861 // Copy the state of a single sequence into the specified buffer862 LLAMA_API size_t llama_state_seq_get_data(863 struct llama_context * ctx,864 uint8_t * dst,865 size_t size,866 llama_seq_id seq_id);867 868 // Copy the sequence data (originally copied with `llama_state_seq_get_data`) into the specified sequence869 // Returns:870 // - Positive: Ok871 // - Zero: Failed to load872 LLAMA_API size_t llama_state_seq_set_data(873 struct llama_context * ctx,874 const uint8_t * src,875 size_t size,876 llama_seq_id dest_seq_id);877 878 LLAMA_API size_t llama_state_seq_save_file(879 struct llama_context * ctx,880 const char * filepath,881 llama_seq_id seq_id,882 const llama_token * tokens,883 size_t n_token_count);884 885 LLAMA_API size_t llama_state_seq_load_file(886 struct llama_context * ctx,887 const char * filepath,888 llama_seq_id dest_seq_id,889 llama_token * tokens_out,890 size_t n_token_capacity,891 size_t * n_token_count_out);892 893#define LLAMA_STATE_SEQ_FLAGS_NONE 0894 895// for backwards-compat896#define LLAMA_STATE_SEQ_FLAGS_SWA_ONLY 1897 898// work only with partial states, such as SWA KV cache or recurrent cache (e.g. Mamba)899#define LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY 1900 901// Keeps the tensor data on device buffers (i.e. not accessible in host memory, but faster save/load).902// Getting the state for a seq_id with this flag invalidates all prior states gotten for that seq_id with this flag.903#define LLAMA_STATE_SEQ_FLAGS_ON_DEVICE 2904 905 typedef uint32_t llama_state_seq_flags;906 907 LLAMA_API size_t llama_state_seq_get_size_ext(908 struct llama_context * ctx,909 llama_seq_id seq_id,910 llama_state_seq_flags flags);911 912 LLAMA_API size_t llama_state_seq_get_data_ext(913 struct llama_context * ctx,914 uint8_t * dst,915 size_t size,916 llama_seq_id seq_id,917 llama_state_seq_flags flags);918 919 LLAMA_API size_t llama_state_seq_set_data_ext(920 struct llama_context * ctx,921 const uint8_t * src,922 size_t size,923 llama_seq_id dest_seq_id,924 llama_state_seq_flags flags);925 926 //927 // Decoding928 //929 930 // Return batch for single sequence of tokens931 // The sequence ID will be fixed to 0932 // The position of the tokens will be tracked automatically by llama_decode933 //934 // NOTE: this is a helper function to facilitate transition to the new batch API - avoid using it935 //936 LLAMA_API struct llama_batch llama_batch_get_one(937 llama_token * tokens,938 int32_t n_tokens);939 940 // Allocates a batch of tokens on the heap that can hold a maximum of n_tokens941 // Each token can be assigned up to n_seq_max sequence ids942 // The batch has to be freed with llama_batch_free()943 // If embd != 0, llama_batch.embd will be allocated with size of n_tokens * embd * sizeof(float)944 // Otherwise, llama_batch.token will be allocated to store n_tokens llama_token945 // The rest of the llama_batch members are allocated with size n_tokens946 // All members are left uninitialized947 LLAMA_API struct llama_batch llama_batch_init(948 int32_t n_tokens,949 int32_t embd,950 int32_t n_seq_max);951 952 // Frees a batch of tokens allocated with llama_batch_init()953 LLAMA_API void llama_batch_free(struct llama_batch batch);954 955 // Process a batch of tokens.956 // In contrast to llama_decode() - this call does not use KV cache.957 // For encode-decoder contexts, processes the batch using the encoder.958 // Can store the encoder output internally for later use by the decoder's cross-attention layers.959 // 0 - success960 // < 0 - error. the memory state is restored to the state before this call961 LLAMA_API int32_t llama_encode(962 struct llama_context * ctx,963 struct llama_batch batch);964 965 // Process a batch of tokens.966 // Requires the context to have a memory.967 // For encode-decoder contexts, processes the batch using the decoder.968 // Positive return values does not mean a fatal error, but rather a warning.969 // Upon fatal-error or abort, the ubatches that managed to be been processed will remain in the memory state of the context970 // To handle this correctly, query the memory state using llama_memory_seq_pos_min() and llama_memory_seq_pos_max()971 // Upon other return values, the memory state is restored to the state before this call972 // 0 - success973 // 1 - could not find a KV slot for the batch (try reducing the size of the batch or increase the context)974 // 2 - aborted (processed ubatches will remain in the context's memory)975 // -1 - invalid input batch976 // < -1 - fatal error (processed ubatches will remain in the context's memory)977 LLAMA_API int32_t llama_decode(978 struct llama_context * ctx,979 struct llama_batch batch);980 981 // Set the number of threads used for decoding982 // n_threads is the number of threads used for generation (single token)983 // n_threads_batch is the number of threads used for prompt and batch processing (multiple tokens)984 LLAMA_API void llama_set_n_threads(struct llama_context * ctx, int32_t n_threads, int32_t n_threads_batch);985 986 // Get the number of threads used for generation of a single token.987 LLAMA_API int32_t llama_n_threads(struct llama_context * ctx);988 989 // Get the number of threads used for prompt and batch processing (multiple token).990 LLAMA_API int32_t llama_n_threads_batch(struct llama_context * ctx);991 992 // Set whether the context outputs embeddings or not993 // TODO: rename to avoid confusion with llama_get_embeddings()994 LLAMA_API void llama_set_embeddings(struct llama_context * ctx, bool embeddings);995 996 // Set whether to use causal attention or not997 // If set to true, the model will only attend to the past tokens998 LLAMA_API void llama_set_causal_attn(struct llama_context * ctx, bool causal_attn);999 1000 // Set whether the model is in warmup mode or not1001 // If true, all model tensors are activated during llama_decode() to load and cache their weights.1002 //1003 // note: using this can cause extra graph reallocations because it changes the graph topology with MoE models,1004 // so it is generally not recommended to use in practice. will be removed in the future1005 DEPRECATED(LLAMA_API void llama_set_warmup(struct llama_context * ctx, bool warmup),1006 "user code should do warmup runs manually [TAG_LLAMA_GRAPH_NO_WARMUP]");1007 1008 // Set abort callback1009 LLAMA_API void llama_set_abort_callback(struct llama_context * ctx, ggml_abort_callback abort_callback, void * abort_callback_data);1010 1011 // Wait until all computations are finished1012 // This is automatically done when using one of the functions below to obtain the computation results1013 // and is not necessary to call it explicitly in most cases1014 LLAMA_API void llama_synchronize(struct llama_context * ctx);1015 1016 // Token logits obtained from the last call to llama_decode()1017 // The logits for which llama_batch.logits[i] != 0 are stored contiguously1018 // in the order they have appeared in the batch.1019 // Rows: number of tokens for which llama_batch.logits[i] != 01020 // Cols: n_vocab1021 // TODO: deprecate in favor of llama_get_logits_ith() (ref: https://github.com/ggml-org/llama.cpp/pull/14853#issuecomment-3113143522)1022 LLAMA_API float * llama_get_logits(struct llama_context * ctx);1023 1024 // Logits for the ith token. For positive indices, Equivalent to:1025 // llama_get_logits(ctx) + ctx->output_ids[i]*n_vocab1026 // Negative indices can be used to access logits in reverse order, -1 is the last logit.1027 // returns NULL for invalid ids.1028 LLAMA_API float * llama_get_logits_ith(struct llama_context * ctx, int32_t i);1029 1030 // Get all output token embeddings.1031 // when pooling_type == LLAMA_POOLING_TYPE_NONE or when using a generative model,1032 // the embeddings for which llama_batch.logits[i] != 0 are stored contiguously1033 // in the order they have appeared in the batch.1034 // shape: [n_outputs*n_embd]1035 // Otherwise, returns NULL.1036 // TODO: deprecate in favor of llama_get_embeddings_ith() (ref: https://github.com/ggml-org/llama.cpp/pull/14853#issuecomment-3113143522)1037 LLAMA_API float * llama_get_embeddings(struct llama_context * ctx);1038 1039 // Get the embeddings for the ith token. For positive indices, Equivalent to:1040 // llama_get_embeddings(ctx) + ctx->output_ids[i]*n_embd1041 // Negative indices can be used to access embeddings in reverse order, -1 is the last embedding.1042 // shape: [n_embd] (1-dimensional)1043 // returns NULL for invalid ids.1044 LLAMA_API float * llama_get_embeddings_ith(struct llama_context * ctx, int32_t i);1045 1046 // Get the embeddings for a sequence id1047 // Returns NULL if pooling_type is LLAMA_POOLING_TYPE_NONE1048 // when pooling_type == LLAMA_POOLING_TYPE_RANK, returns float[n_cls_out] with the rank(s) of the sequence1049 // otherwise: float[n_embd] (1-dimensional)1050 LLAMA_API float * llama_get_embeddings_seq(struct llama_context * ctx, llama_seq_id seq_id);1051 1052 //1053 // backend sampling API [EXPERIMENTAL]1054 // note: use only if the llama_context was created with at least one llama_sampler_seq_config1055 //1056 1057 // Get the backend sampled token for the ith token.1058 // With multiple outputs, sampler state advances when the token is accepted,1059 // not when it is read through this function.1060 // When accepting multiple outputs, accept a contiguous prefix in output order.1061 // Returns LLAMA_TOKEN_NULL if no token was sampled.1062 LLAMA_API llama_token llama_get_sampled_token_ith(struct llama_context * ctx, int32_t i);1063 1064 // Get the backend sampled probabilities for the ith token1065 // The index matches llama_get_sampled_token_ith().1066 // Returns NULL if no probabilities were generated.1067 LLAMA_API float * llama_get_sampled_probs_ith (struct llama_context * ctx, int32_t i);1068 LLAMA_API uint32_t llama_get_sampled_probs_count_ith(struct llama_context * ctx, int32_t i);1069 1070 // Get the backend sampled logits for the ith token1071 // Returns NULL if no logits were sampled.1072 LLAMA_API float * llama_get_sampled_logits_ith (struct llama_context * ctx, int32_t i);1073 LLAMA_API uint32_t llama_get_sampled_logits_count_ith(struct llama_context * ctx, int32_t i);1074 1075 // Get the backend sampled candidates (token ids) for the ith token1076 // These are needed to map probability/logit indices to vocab token ids.1077 // Returns NULL if no candidates were sampled.1078 LLAMA_API llama_token * llama_get_sampled_candidates_ith (struct llama_context * ctx, int32_t i);1079 LLAMA_API uint32_t llama_get_sampled_candidates_count_ith(struct llama_context * ctx, int32_t i);1080 1081 //1082 // Vocab1083 //1084 1085 LLAMA_API const char * llama_vocab_get_text(const struct llama_vocab * vocab, llama_token token);1086 1087 LLAMA_API float llama_vocab_get_score(const struct llama_vocab * vocab, llama_token token);1088 1089 LLAMA_API enum llama_token_attr llama_vocab_get_attr(const struct llama_vocab * vocab, llama_token token);1090 1091 // Check if the token is supposed to end generation (end-of-generation, eg. EOS, EOT, etc.)1092 LLAMA_API bool llama_vocab_is_eog(const struct llama_vocab * vocab, llama_token token);1093 1094 // Identify if Token Id is a control token or a render-able token1095 LLAMA_API bool llama_vocab_is_control(const struct llama_vocab * vocab, llama_token token);1096 1097 // Special tokens1098 LLAMA_API llama_token llama_vocab_bos(const struct llama_vocab * vocab); // beginning-of-sentence1099 LLAMA_API llama_token llama_vocab_eos(const struct llama_vocab * vocab); // end-of-sentence1100 LLAMA_API llama_token llama_vocab_eot(const struct llama_vocab * vocab); // end-of-turn1101 LLAMA_API llama_token llama_vocab_sep(const struct llama_vocab * vocab); // sentence separator1102 LLAMA_API llama_token llama_vocab_nl (const struct llama_vocab * vocab); // next-line1103 LLAMA_API llama_token llama_vocab_pad(const struct llama_vocab * vocab); // padding1104 LLAMA_API llama_token llama_vocab_mask(const struct llama_vocab * vocab); // mask1105 1106 LLAMA_API bool llama_vocab_get_add_bos(const struct llama_vocab * vocab);1107 LLAMA_API bool llama_vocab_get_add_eos(const struct llama_vocab * vocab);1108 LLAMA_API bool llama_vocab_get_add_sep(const struct llama_vocab * vocab);1109 1110 // model-specific suppress tokens (gguf key: tokenizer.ggml.suppress_tokens)1111 LLAMA_API const llama_token * llama_vocab_get_suppress_tokens(const struct llama_vocab * vocab, int32_t * n_suppress_tokens);1112 1113 LLAMA_API llama_token llama_vocab_fim_pre(const struct llama_vocab * vocab);1114 LLAMA_API llama_token llama_vocab_fim_suf(const struct llama_vocab * vocab);1115 LLAMA_API llama_token llama_vocab_fim_mid(const struct llama_vocab * vocab);1116 LLAMA_API llama_token llama_vocab_fim_pad(const struct llama_vocab * vocab);1117 LLAMA_API llama_token llama_vocab_fim_rep(const struct llama_vocab * vocab);1118 LLAMA_API llama_token llama_vocab_fim_sep(const struct llama_vocab * vocab);1119 1120 DEPRECATED(LLAMA_API const char * llama_token_get_text(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_get_text instead");1121 DEPRECATED(LLAMA_API float llama_token_get_score(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_get_score instead");1122 DEPRECATED(LLAMA_API enum llama_token_attr llama_token_get_attr(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_get_attr instead");1123 DEPRECATED(LLAMA_API bool llama_token_is_eog(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_is_eog instead");1124 DEPRECATED(LLAMA_API bool llama_token_is_control(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_is_control instead");1125 DEPRECATED(LLAMA_API llama_token llama_token_bos(const struct llama_vocab * vocab), "use llama_vocab_bos instead");1126 DEPRECATED(LLAMA_API llama_token llama_token_eos(const struct llama_vocab * vocab), "use llama_vocab_eos instead");1127 DEPRECATED(LLAMA_API llama_token llama_token_eot(const struct llama_vocab * vocab), "use llama_vocab_eot instead");1128 DEPRECATED(LLAMA_API llama_token llama_token_cls(const struct llama_vocab * vocab), "use llama_vocab_cls instead");1129 DEPRECATED(LLAMA_API llama_token llama_token_sep(const struct llama_vocab * vocab), "use llama_vocab_sep instead");1130 DEPRECATED(LLAMA_API llama_token llama_token_nl (const struct llama_vocab * vocab), "use llama_vocab_nl instead");1131 DEPRECATED(LLAMA_API llama_token llama_token_pad(const struct llama_vocab * vocab), "use llama_vocab_pad instead");1132 DEPRECATED(LLAMA_API bool llama_add_bos_token(const struct llama_vocab * vocab), "use llama_vocab_get_add_bos instead");1133 DEPRECATED(LLAMA_API bool llama_add_eos_token(const struct llama_vocab * vocab), "use llama_vocab_get_add_eos instead");1134 DEPRECATED(LLAMA_API llama_token llama_token_fim_pre(const struct llama_vocab * vocab), "use llama_vocab_fim_pre instead");1135 DEPRECATED(LLAMA_API llama_token llama_token_fim_suf(const struct llama_vocab * vocab), "use llama_vocab_fim_suf instead");1136 DEPRECATED(LLAMA_API llama_token llama_token_fim_mid(const struct llama_vocab * vocab), "use llama_vocab_fim_mid instead");1137 DEPRECATED(LLAMA_API llama_token llama_token_fim_pad(const struct llama_vocab * vocab), "use llama_vocab_fim_pad instead");1138 DEPRECATED(LLAMA_API llama_token llama_token_fim_rep(const struct llama_vocab * vocab), "use llama_vocab_fim_rep instead");1139 DEPRECATED(LLAMA_API llama_token llama_token_fim_sep(const struct llama_vocab * vocab), "use llama_vocab_fim_sep instead");1140 1141 // CLS is equivalent to BOS1142 DEPRECATED(LLAMA_API llama_token llama_vocab_cls(const struct llama_vocab * vocab), // classification1143 "use llama_vocab_bos instead");1144 1145 //1146 // Tokenization1147 //1148 // The API is thread-safe.1149 //1150 1151 /// @details Convert the provided text into tokens.1152 /// @param tokens The tokens pointer must be large enough to hold the resulting tokens.1153 /// @return Returns the number of tokens on success, no more than n_tokens_max1154 /// @return Returns a negative number on failure - the number of tokens that would have been returned1155 /// @return Returns INT32_MIN on overflow (e.g., tokenization result size exceeds int32_t limit)1156 /// @param add_special Allow to add BOS and EOS tokens if model is configured to do so.1157 /// @param parse_special Allow tokenizing special and/or control tokens which otherwise are not exposed and treated1158 /// as plaintext. Does not insert a leading space.1159 LLAMA_API int32_t llama_tokenize(1160 const struct llama_vocab * vocab,1161 const char * text,1162 int32_t text_len,1163 llama_token * tokens,1164 int32_t n_tokens_max,1165 bool add_special,1166 bool parse_special);1167 1168 // Token Id -> Piece.1169 // Uses the vocabulary in the provided context.1170 // Does not write null terminator to the buffer.1171 // User can skip up to 'lstrip' leading spaces before copying (useful when encoding/decoding multiple tokens with 'add_space_prefix')1172 // @param special If true, special tokens are rendered in the output.1173 LLAMA_API int32_t llama_token_to_piece(1174 const struct llama_vocab * vocab,1175 llama_token token,1176 char * buf,1177 int32_t length,1178 int32_t lstrip,1179 bool special);1180 1181 /// @details Convert the provided tokens into text (inverse of llama_tokenize()).1182 /// @param text The char pointer must be large enough to hold the resulting text.1183 /// @return Returns the number of chars/bytes on success, no more than text_len_max.1184 /// @return Returns a negative number on failure - the number of chars/bytes that would have been returned.1185 /// @param remove_special Allow to remove BOS and EOS tokens if model is configured to do so.1186 /// @param unparse_special If true, special tokens are rendered in the output.1187 LLAMA_API int32_t llama_detokenize(1188 const struct llama_vocab * vocab,1189 const llama_token * tokens,1190 int32_t n_tokens,1191 char * text,1192 int32_t text_len_max,1193 bool remove_special,1194 bool unparse_special);1195 1196 //1197 // Chat templates1198 //1199 1200 /// Apply chat template. Inspired by hf apply_chat_template() on python.