Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
llama.h1626 linesDownload Raw Back to include
1#ifndef LLAMA_H2#define LLAMA_H3 4#include "ggml.h"5#include "ggml-cpu.h"6#include "ggml-backend.h"7#include "ggml-opt.h"8#include "gguf.h"9 10#include <stddef.h>11#include <stdint.h>12#include <stdio.h>13#include <stdbool.h>14 15#ifdef LLAMA_SHARED16#    if defined(_WIN32) && !defined(__MINGW32__)17#        ifdef LLAMA_BUILD18#            define LLAMA_API __declspec(dllexport)19#        else20#            define LLAMA_API __declspec(dllimport)21#        endif22#    else23#        define LLAMA_API __attribute__ ((visibility ("default")))24#    endif25#else26#    define LLAMA_API27#endif28 29#ifdef __GNUC__30#    define DEPRECATED(func, hint) func __attribute__((deprecated(hint)))31#elif defined(_MSC_VER)32#    define DEPRECATED(func, hint) __declspec(deprecated(hint)) func33#else34#    define DEPRECATED(func, hint) func35#endif36 37#define LLAMA_DEFAULT_SEED 0xFFFFFFFF38 39#define LLAMA_TOKEN_NULL -140 41#define LLAMA_FILE_MAGIC_GGLA 0x67676c61u // 'ggla'42#define LLAMA_FILE_MAGIC_GGSN 0x6767736eu // 'ggsn'43#define LLAMA_FILE_MAGIC_GGSQ 0x67677371u // 'ggsq'44 45#define LLAMA_SESSION_MAGIC   LLAMA_FILE_MAGIC_GGSN46#define LLAMA_SESSION_VERSION 947 48#define LLAMA_STATE_SEQ_MAGIC   LLAMA_FILE_MAGIC_GGSQ49#define LLAMA_STATE_SEQ_VERSION 250 51#ifdef __cplusplus52extern "C" {53#endif54 55    //56    // C interface57    //58    // TODO: show sample usage59    //60 61    struct llama_vocab;62    struct llama_model;63    struct llama_context;64    struct llama_sampler;65 66    typedef struct llama_memory_i * llama_memory_t;67 68    typedef int32_t llama_pos;69    typedef int32_t llama_token;70    typedef int32_t llama_seq_id;71 72    enum llama_vocab_type {73        LLAMA_VOCAB_TYPE_NONE   = 0, // For models without vocab74        LLAMA_VOCAB_TYPE_SPM    = 1, // LLaMA tokenizer based on byte-level BPE with byte fallback75        LLAMA_VOCAB_TYPE_BPE    = 2, // GPT-2 tokenizer based on byte-level BPE76        LLAMA_VOCAB_TYPE_WPM    = 3, // BERT tokenizer based on WordPiece77        LLAMA_VOCAB_TYPE_UGM    = 4, // T5 tokenizer based on Unigram78        LLAMA_VOCAB_TYPE_RWKV   = 5, // RWKV tokenizer based on greedy tokenization79        LLAMA_VOCAB_TYPE_PLAMO2 = 6, // PLaMo-2 tokenizer based on Aho-Corasick with dynamic programming80    };81 82    enum llama_rope_type {83        LLAMA_ROPE_TYPE_NONE   = -1,84        LLAMA_ROPE_TYPE_NORM   = 0,85        LLAMA_ROPE_TYPE_NEOX   = GGML_ROPE_TYPE_NEOX,86        LLAMA_ROPE_TYPE_MROPE  = GGML_ROPE_TYPE_MROPE,87        LLAMA_ROPE_TYPE_IMROPE = GGML_ROPE_TYPE_IMROPE,88        LLAMA_ROPE_TYPE_VISION = GGML_ROPE_TYPE_VISION,89    };90 91    enum llama_token_type { //TODO: remove, required until per token attributes are available from GGUF file92        LLAMA_TOKEN_TYPE_UNDEFINED    = 0,93        LLAMA_TOKEN_TYPE_NORMAL       = 1,94        LLAMA_TOKEN_TYPE_UNKNOWN      = 2,95        LLAMA_TOKEN_TYPE_CONTROL      = 3,96        LLAMA_TOKEN_TYPE_USER_DEFINED = 4,97        LLAMA_TOKEN_TYPE_UNUSED       = 5,98        LLAMA_TOKEN_TYPE_BYTE         = 6,99    };100 101    enum llama_token_attr {102        LLAMA_TOKEN_ATTR_UNDEFINED    = 0,103        LLAMA_TOKEN_ATTR_UNKNOWN      = 1 << 0,104        LLAMA_TOKEN_ATTR_UNUSED       = 1 << 1,105        LLAMA_TOKEN_ATTR_NORMAL       = 1 << 2,106        LLAMA_TOKEN_ATTR_CONTROL      = 1 << 3,  // SPECIAL?107        LLAMA_TOKEN_ATTR_USER_DEFINED = 1 << 4,108        LLAMA_TOKEN_ATTR_BYTE         = 1 << 5,109        LLAMA_TOKEN_ATTR_NORMALIZED   = 1 << 6,110        LLAMA_TOKEN_ATTR_LSTRIP       = 1 << 7,111        LLAMA_TOKEN_ATTR_RSTRIP       = 1 << 8,112        LLAMA_TOKEN_ATTR_SINGLE_WORD  = 1 << 9,113    };114 115    // model file types116    enum llama_ftype {117        LLAMA_FTYPE_ALL_F32              = 0,118        LLAMA_FTYPE_MOSTLY_F16           = 1,  // except 1d tensors119        LLAMA_FTYPE_MOSTLY_Q4_0          = 2,  // except 1d tensors120        LLAMA_FTYPE_MOSTLY_Q4_1          = 3,  // except 1d tensors121        // LLAMA_FTYPE_MOSTLY_Q4_1_SOME_F16 = 4,  // tok_embeddings.weight and output.weight are F16122        // LLAMA_FTYPE_MOSTLY_Q4_2       = 5,  // support has been removed123        // LLAMA_FTYPE_MOSTLY_Q4_3       = 6,  // support has been removed124        LLAMA_FTYPE_MOSTLY_Q8_0          = 7,  // except 1d tensors125        LLAMA_FTYPE_MOSTLY_Q5_0          = 8,  // except 1d tensors126        LLAMA_FTYPE_MOSTLY_Q5_1          = 9,  // except 1d tensors127        LLAMA_FTYPE_MOSTLY_Q2_K          = 10, // except 1d tensors128        LLAMA_FTYPE_MOSTLY_Q3_K_S        = 11, // except 1d tensors129        LLAMA_FTYPE_MOSTLY_Q3_K_M        = 12, // except 1d tensors130        LLAMA_FTYPE_MOSTLY_Q3_K_L        = 13, // except 1d tensors131        LLAMA_FTYPE_MOSTLY_Q4_K_S        = 14, // except 1d tensors132        LLAMA_FTYPE_MOSTLY_Q4_K_M        = 15, // except 1d tensors133        LLAMA_FTYPE_MOSTLY_Q5_K_S        = 16, // except 1d tensors134        LLAMA_FTYPE_MOSTLY_Q5_K_M        = 17, // except 1d tensors135        LLAMA_FTYPE_MOSTLY_Q6_K          = 18, // except 1d tensors136        LLAMA_FTYPE_MOSTLY_IQ2_XXS       = 19, // except 1d tensors137        LLAMA_FTYPE_MOSTLY_IQ2_XS        = 20, // except 1d tensors138        LLAMA_FTYPE_MOSTLY_Q2_K_S        = 21, // except 1d tensors139        LLAMA_FTYPE_MOSTLY_IQ3_XS        = 22, // except 1d tensors140        LLAMA_FTYPE_MOSTLY_IQ3_XXS       = 23, // except 1d tensors141        LLAMA_FTYPE_MOSTLY_IQ1_S         = 24, // except 1d tensors142        LLAMA_FTYPE_MOSTLY_IQ4_NL        = 25, // except 1d tensors143        LLAMA_FTYPE_MOSTLY_IQ3_S         = 26, // except 1d tensors144        LLAMA_FTYPE_MOSTLY_IQ3_M         = 27, // except 1d tensors145        LLAMA_FTYPE_MOSTLY_IQ2_S         = 28, // except 1d tensors146        LLAMA_FTYPE_MOSTLY_IQ2_M         = 29, // except 1d tensors147        LLAMA_FTYPE_MOSTLY_IQ4_XS        = 30, // except 1d tensors148        LLAMA_FTYPE_MOSTLY_IQ1_M         = 31, // except 1d tensors149        LLAMA_FTYPE_MOSTLY_BF16          = 32, // except 1d tensors150        //LLAMA_FTYPE_MOSTLY_Q4_0_4_4      = 33, // removed from gguf files, use Q4_0 and runtime repack151        //LLAMA_FTYPE_MOSTLY_Q4_0_4_8      = 34, // removed from gguf files, use Q4_0 and runtime repack152        //LLAMA_FTYPE_MOSTLY_Q4_0_8_8      = 35, // removed from gguf files, use Q4_0 and runtime repack153        LLAMA_FTYPE_MOSTLY_TQ1_0         = 36, // except 1d tensors154        LLAMA_FTYPE_MOSTLY_TQ2_0         = 37, // except 1d tensors155        LLAMA_FTYPE_MOSTLY_MXFP4_MOE     = 38, // except 1d tensors156        LLAMA_FTYPE_MOSTLY_NVFP4         = 39, // except 1d tensors157        LLAMA_FTYPE_MOSTLY_Q1_0          = 40, // except 1d tensors158        LLAMA_FTYPE_MOSTLY_Q2_0          = 41, // except 1d tensors159 160        LLAMA_FTYPE_GUESSED = 1024, // not specified in the model file161    };162 163    // Get the model file type (quantization) as a string, e.g. "Q8_0" or "Q4_K - Medium"164    LLAMA_API const char * llama_ftype_name(enum llama_ftype ftype);165 166    enum llama_rope_scaling_type {167        LLAMA_ROPE_SCALING_TYPE_UNSPECIFIED = -1,168        LLAMA_ROPE_SCALING_TYPE_NONE        = 0,169        LLAMA_ROPE_SCALING_TYPE_LINEAR      = 1,170        LLAMA_ROPE_SCALING_TYPE_YARN        = 2,171        LLAMA_ROPE_SCALING_TYPE_LONGROPE    = 3,172        LLAMA_ROPE_SCALING_TYPE_MAX_VALUE   = LLAMA_ROPE_SCALING_TYPE_LONGROPE,173    };174 175    enum llama_pooling_type {176        LLAMA_POOLING_TYPE_UNSPECIFIED = -1,177        LLAMA_POOLING_TYPE_NONE = 0,178        LLAMA_POOLING_TYPE_MEAN = 1,179        LLAMA_POOLING_TYPE_CLS  = 2,180        LLAMA_POOLING_TYPE_LAST = 3,181        LLAMA_POOLING_TYPE_RANK = 4, // used by reranking models to attach the classification head to the graph182    };183 184    enum llama_attention_type {185        LLAMA_ATTENTION_TYPE_UNSPECIFIED = -1,186        LLAMA_ATTENTION_TYPE_CAUSAL      = 0,187        LLAMA_ATTENTION_TYPE_NON_CAUSAL  = 1,188    };189 190    enum llama_flash_attn_type {191        LLAMA_FLASH_ATTN_TYPE_AUTO     = -1,192        LLAMA_FLASH_ATTN_TYPE_DISABLED = 0,193        LLAMA_FLASH_ATTN_TYPE_ENABLED  = 1,194    };195 196    LLAMA_API const char * llama_flash_attn_type_name(enum llama_flash_attn_type flash_attn_type);197 198    enum llama_split_mode {199        LLAMA_SPLIT_MODE_NONE   = 0, // single GPU200        LLAMA_SPLIT_MODE_LAYER  = 1, // split layers and KV across GPUs201        LLAMA_SPLIT_MODE_ROW    = 2, // split layers and KV across GPUs, use tensor parallelism if supported202        LLAMA_SPLIT_MODE_TENSOR = 3,203    };204 205    enum llama_load_mode {206        LLAMA_LOAD_MODE_NONE       = 0, // no special loading mode207        LLAMA_LOAD_MODE_MMAP       = 1, // memory map the model208        LLAMA_LOAD_MODE_MLOCK      = 2, // force system to keep model in RAM rather than swapping or compressing209        LLAMA_LOAD_MODE_MMAP_MLOCK = 3, // mmap + force system to keep model in RAM rather than swapping or compressing210        LLAMA_LOAD_MODE_DIRECT_IO  = 4, // use direct I/O if available211    };212 213    LLAMA_API const char * llama_load_mode_name(enum llama_load_mode load_mode);214    LLAMA_API enum llama_load_mode llama_load_mode_from_str(const char * str);215 216    enum llama_context_type {217        LLAMA_CONTEXT_TYPE_DEFAULT = 0,218        LLAMA_CONTEXT_TYPE_MTP     = 1,219    };220 221    // TODO: simplify (https://github.com/ggml-org/llama.cpp/pull/9294#pullrequestreview-2286561979)222    typedef struct llama_token_data {223        llama_token id; // token id224        float logit;    // log-odds of the token225        float p;        // probability of the token226    } llama_token_data;227 228    typedef struct llama_token_data_array {229        // TODO: consider SoA230        // NOTE: this pointer can be modified by the samplers231        llama_token_data * data;232        size_t size;233        int64_t selected; // this is the index in the data array (i.e. not the token id)234        bool sorted;      // note: do not assume the data is sorted - always check this flag235    } llama_token_data_array;236 237    typedef bool (*llama_progress_callback)(float progress, void * user_data);238 239    // Input data for llama_encode/llama_decode240    // A llama_batch object can contain input about one or many sequences241    // The provided arrays (i.e. token, embd, pos, etc.) must have size of n_tokens242    //243    // - token  : the token ids of the input (used when embd is NULL)244    // - embd   : token embeddings (i.e. float vector of size n_embd) (used when token is NULL)245    // - pos    : the positions of the respective token in the sequence246    //            (if set to NULL, the token position will be tracked automatically by llama_encode/llama_decode)247    // - seq_id : the sequence to which the respective token belongs248    //            (if set to NULL, the sequence ID will be assumed to be 0)249    // - logits : if zero, the logits (and/or the embeddings) for the respective token will not be output250    //            (if set to NULL:251    //               - if embeddings: all tokens are output252    //               - if not:        only the last token is output253    //            )254    //255    typedef struct llama_batch {256        int32_t n_tokens;257 258        llama_token  *  token;259        float        *  embd;260        llama_pos    *  pos;261        int32_t      *  n_seq_id;262        llama_seq_id ** seq_id;263        int8_t       *  logits;   // TODO: rename this to "output"264    } llama_batch;265 266    enum llama_model_kv_override_type {267        LLAMA_KV_OVERRIDE_TYPE_INT,268        LLAMA_KV_OVERRIDE_TYPE_FLOAT,269        LLAMA_KV_OVERRIDE_TYPE_BOOL,270        LLAMA_KV_OVERRIDE_TYPE_STR,271    };272 273    enum llama_model_meta_key {274        LLAMA_MODEL_META_KEY_SAMPLING_SEQUENCE,275        LLAMA_MODEL_META_KEY_SAMPLING_TOP_K,276        LLAMA_MODEL_META_KEY_SAMPLING_TOP_P,277        LLAMA_MODEL_META_KEY_SAMPLING_MIN_P,278        LLAMA_MODEL_META_KEY_SAMPLING_XTC_PROBABILITY,279        LLAMA_MODEL_META_KEY_SAMPLING_XTC_THRESHOLD,280        LLAMA_MODEL_META_KEY_SAMPLING_TEMP,281        LLAMA_MODEL_META_KEY_SAMPLING_PENALTY_LAST_N,282        LLAMA_MODEL_META_KEY_SAMPLING_PENALTY_REPEAT,283        LLAMA_MODEL_META_KEY_SAMPLING_MIROSTAT,284        LLAMA_MODEL_META_KEY_SAMPLING_MIROSTAT_TAU,285        LLAMA_MODEL_META_KEY_SAMPLING_MIROSTAT_ETA,286    };287 288    struct llama_model_kv_override {289        enum llama_model_kv_override_type tag;290 291        char key[128];292 293        union {294            int64_t val_i64;295            double  val_f64;296            bool    val_bool;297            char    val_str[128];298        };299    };300 301    struct llama_model_tensor_buft_override {302        const char * pattern;303        ggml_backend_buffer_type_t buft;304    };305 306    struct llama_model_params {307        // NULL-terminated list of devices to use for offloading (if NULL, all available devices are used)308        ggml_backend_dev_t * devices;309 310        // NULL-terminated list of buffer types to use for tensors that match a pattern311        const struct llama_model_tensor_buft_override * tensor_buft_overrides;312 313        int32_t n_gpu_layers; // number of layers to store in VRAM, a negative value means all layers314        enum llama_split_mode split_mode; // how to split the model across multiple GPUs315        enum llama_load_mode  load_mode;  // how to load the model316 317        // the GPU that is used for the entire model when split_mode is LLAMA_SPLIT_MODE_NONE318        int32_t main_gpu;319 320        // proportion of the model (layers or rows) to offload to each GPU, size: llama_max_devices()321        const float * tensor_split;322 323        // Called with a progress value between 0.0 and 1.0. Pass NULL to disable.324        // If the provided progress_callback returns true, model loading continues.325        // If it returns false, model loading is immediately aborted.326        llama_progress_callback progress_callback;327 328        // context pointer passed to the progress callback329        void * progress_callback_user_data;330 331        // override key-value pairs of the model meta data332        const struct llama_model_kv_override * kv_overrides;333 334        // Keep the booleans together to avoid misalignment during copy-by-value.335        bool vocab_only;      // only load the vocabulary, no weights336        bool check_tensors;   // validate model tensor data337        bool use_extra_bufts; // use extra buffer types (used for weight repacking)338        bool no_host;         // bypass host buffer allowing extra buffers to be used339        bool no_alloc;        // only load metadata and simulate memory allocations340        bool load_mtp;        // whether to load MTP layers341    };342 343    struct llama_sampler_seq_config {344        llama_seq_id           seq_id;345        struct llama_sampler * sampler;346    };347 348    // NOTE: changing the default values of parameters marked as [EXPERIMENTAL] may cause crashes or incorrect results in certain configurations349    //       https://github.com/ggml-org/llama.cpp/pull/7544350    struct llama_context_params {351        uint32_t n_ctx;                 // text context, 0 = from model352        uint32_t n_batch;               // logical maximum batch size that can be submitted to llama_decode353        uint32_t n_ubatch;              // physical maximum batch size354        uint32_t n_seq_max;             // max number of sequences (i.e. distinct states for recurrent models)355        uint32_t n_rs_seq;              // number of recurrent-state snapshots per seq for rollback (0 = no rollback) [EXPERIMENTAL]356        uint32_t n_outputs_max;         // max outputs in a ubatch (0 = n_batch)357        uint32_t n_outputs_max_per_seq; // max outputs per sequence (0 = n_outputs_max)358        int32_t  n_threads;             // number of threads to use for generation359        int32_t  n_threads_batch;       // number of threads to use for batch processing360 361        enum llama_context_type      ctx_type;          // set the context type (e.g. MTP)362        enum llama_rope_scaling_type rope_scaling_type; // RoPE scaling type, from `enum llama_rope_scaling_type`363        enum llama_pooling_type      pooling_type;      // whether to pool (sum) embedding results by sequence id364        enum llama_attention_type    attention_type;    // attention type to use for embeddings365        enum llama_flash_attn_type   flash_attn_type;   // when to enable Flash Attention366 367        // ref: https://github.com/ggml-org/llama.cpp/pull/2054368        float    rope_freq_base;   // RoPE base frequency, 0 = from model369        float    rope_freq_scale;  // RoPE frequency scaling factor, 0 = from model370        float    yarn_ext_factor;  // YaRN extrapolation mix factor, negative = from model371        float    yarn_attn_factor; // YaRN magnitude scaling factor372        float    yarn_beta_fast;   // YaRN low correction dim373        float    yarn_beta_slow;   // YaRN high correction dim374        uint32_t yarn_orig_ctx;    // YaRN original context size375        float    defrag_thold;     // [DEPRECATED] defragment the KV cache if holes/size > thold, <= 0 disabled (default)376 377        ggml_backend_sched_eval_callback cb_eval;378        void * cb_eval_user_data;379 380        enum ggml_type type_k; // data type for K cache [EXPERIMENTAL]381        enum ggml_type type_v; // data type for V cache [EXPERIMENTAL]382 383        // Abort callback384        // if it returns true, execution of llama_decode() will be aborted385        // currently works only with CPU execution386        ggml_abort_callback abort_callback;387        void *              abort_callback_data;388 389        // Keep the booleans together and at the end of the struct to avoid misalignment during copy-by-value.390        bool embeddings;  // if true, extract embeddings (together with logits)391        bool offload_kqv; // offload the KQV ops (including the KV cache) to GPU392        bool no_perf;     // measure performance timings393        bool op_offload;  // offload host tensor operations to device394        bool swa_full;    // use full-size SWA cache (https://github.com/ggml-org/llama.cpp/pull/13194#issuecomment-2868343055)395                          // NOTE: setting to false when n_seq_max > 1 can cause bad performance in some cases396                          //       ref: https://github.com/ggml-org/llama.cpp/pull/13845#issuecomment-2924800573397        bool kv_unified;  // use a unified buffer across the input sequences when computing the attention398                          // try to disable when n_seq_max > 1 for improved performance when the sequences do not share a large prefix399                          // ref: https://github.com/ggml-org/llama.cpp/pull/14363400 401        // [EXPERIMENTAL]402        // backend sampler chain configuration (make sure the caller keeps the sampler chains alive)403        // note: the samplers must be sampler chains (i.e. use llama_sampler_chain_init)404        struct llama_sampler_seq_config * samplers;405        size_t                            n_samplers;406 407        // a source/target/parent context408        // can be utilized in various ways, for example by sharing results or llama_memory between 2 contexts409        struct llama_context * ctx_other;410    };411 412    struct llama_model_tensor_override {413        const char * pattern;414        enum ggml_type type;415    };416 417    struct llama_model_imatrix_data {418        const char * name;419        const float * data;420        size_t size;421    };422 423    // model quantization parameters424    typedef struct llama_model_quantize_params {425        int32_t nthread;                                            // number of threads to use for quantizing, if <=0 will use std::thread::hardware_concurrency()426        enum llama_ftype ftype;                                     // quantize to this llama_ftype427        enum ggml_type output_tensor_type;                          // output tensor type428        enum ggml_type token_embedding_type;                        // token embeddings tensor type429        bool allow_requantize;                                      // allow quantizing non-f32/f16 tensors430        bool quantize_output_tensor;                                // quantize output.weight431        bool only_copy;                                             // only copy tensors - ftype, allow_requantize and quantize_output_tensor are ignored432        bool pure;                                                  // quantize all tensors to the default type433        bool keep_split;                                            // quantize to the same number of shards434        bool dry_run;                                               // calculate and show the final quantization size without performing quantization435        const struct llama_model_imatrix_data * imatrix;            // pointer to importance matrix data436        const struct llama_model_kv_override * kv_overrides;        // pointer to kv overrides437        const struct llama_model_tensor_override * tt_overrides;    // pointer to tensor overrides438        const int32_t * prune_layers;                               // pointer to layer indices to prune439    } llama_model_quantize_params;440 441    typedef struct llama_logit_bias {442        llama_token token;443        float bias;444    } llama_logit_bias;445 446    typedef struct llama_sampler_chain_params {447        bool no_perf; // whether to measure performance timings448    } llama_sampler_chain_params;449 450    // used in chat template451    typedef struct llama_chat_message {452        const char * role;453        const char * content;454    } llama_chat_message;455 456    // lora adapter457    struct llama_adapter_lora;458 459    // Helpers for getting default parameters460    // TODO: update API to start accepting pointers to params structs (https://github.com/ggml-org/llama.cpp/discussions/9172)461    LLAMA_API struct llama_model_params          llama_model_default_params(void);462    LLAMA_API struct llama_context_params        llama_context_default_params(void);463    LLAMA_API struct llama_sampler_chain_params  llama_sampler_chain_default_params(void);464    LLAMA_API struct llama_model_quantize_params llama_model_quantize_default_params(void);465 466    // Initialize the llama + ggml backend467    // If numa is true, use NUMA optimizations468    // Call once at the start of the program469    LLAMA_API void llama_backend_init(void);470 471    // Call once at the end of the program - currently only used for MPI472    LLAMA_API void llama_backend_free(void);473 474    //optional:475    LLAMA_API void llama_numa_init(enum ggml_numa_strategy numa);476 477    // Optional: an auto threadpool gets created in ggml if not passed explicitly478    LLAMA_API void llama_attach_threadpool(479            struct llama_context * ctx,480               ggml_threadpool_t   threadpool,481               ggml_threadpool_t   threadpool_batch);482 483    LLAMA_API void llama_detach_threadpool(struct llama_context * ctx);484 485    typedef void (*llama_model_set_tensor_data_t)(struct ggml_tensor * tensor, void * userdata);486 487    // Create a new model from GGUF metadata as well as a function to set the tensor data488    //   - tensors are created as GGML_TYPE_F32 by default,489    //     override by adding a tensor with the same name but a different name to the context490    LLAMA_API struct llama_model * llama_model_init_from_user(491                    struct gguf_context * metadata,492          llama_model_set_tensor_data_t   set_tensor_data,    // function to initialize tensor data with493                                   void * set_tensor_data_ud, // userdata for function494              struct llama_model_params   params);495 496    DEPRECATED(LLAMA_API struct llama_model * llama_load_model_from_file(497                             const char * path_model,498              struct llama_model_params   params),499            "use llama_model_load_from_file instead");500 501    // Load a model from a file502    // If the file is split into multiple parts, the file name must follow this pattern: <name>-%05d-of-%05d.gguf503    // If the split file name does not follow this pattern, use llama_model_load_from_splits504    LLAMA_API struct llama_model * llama_model_load_from_file(505                             const char * path_model,506              struct llama_model_params   params);507 508    // Load a model from an open FILE pointer509    LLAMA_API struct llama_model * llama_model_load_from_file_ptr(510                                   FILE * file,511              struct llama_model_params   params);512 513    // Load a model from multiple splits (support custom naming scheme)514    // The paths must be in the correct order515    LLAMA_API struct llama_model * llama_model_load_from_splits(516                             const char ** paths,517                                 size_t    n_paths,518              struct llama_model_params    params);519 520    LLAMA_API void llama_model_save_to_file(521            const struct llama_model * model,522                        const char * path_model);523 524    DEPRECATED(LLAMA_API void llama_free_model(struct llama_model * model),525            "use llama_model_free instead");526 527    LLAMA_API void llama_model_free(struct llama_model * model);528 529    LLAMA_API struct llama_context * llama_init_from_model(530                     struct llama_model * model,531            struct llama_context_params   params);532 533    DEPRECATED(LLAMA_API struct llama_context * llama_new_context_with_model(534                     struct llama_model * model,535            struct llama_context_params   params),536            "use llama_init_from_model instead");537 538    // Frees all allocated memory539    LLAMA_API void llama_free(struct llama_context * ctx);540 541    LLAMA_API int64_t llama_time_us(void);542 543    LLAMA_API size_t llama_max_devices(void);544    LLAMA_API size_t llama_max_parallel_sequences(void);545    LLAMA_API size_t llama_max_tensor_buft_overrides(void);546 547    LLAMA_API bool llama_supports_mmap       (void);548    LLAMA_API bool llama_supports_mlock      (void);549    LLAMA_API bool llama_supports_gpu_offload(void);550    LLAMA_API bool llama_supports_rpc        (void);551 552    // NOTE: After creating a llama_context, it is recommended to query the actual values using these functions553    //       In some cases the requested values via llama_context_params may differ from the actual values used by the context554    //       ref: https://github.com/ggml-org/llama.cpp/pull/17046#discussion_r2503085732555    LLAMA_API uint32_t llama_n_ctx      (const struct llama_context * ctx);556    LLAMA_API uint32_t llama_n_ctx_seq  (const struct llama_context * ctx);557    LLAMA_API uint32_t llama_n_batch    (const struct llama_context * ctx);558    LLAMA_API uint32_t llama_n_ubatch   (const struct llama_context * ctx);559    LLAMA_API uint32_t llama_n_seq_max  (const struct llama_context * ctx);560    LLAMA_API uint32_t llama_n_rs_seq   (const struct llama_context * ctx);561 562    DEPRECATED(LLAMA_API int32_t llama_n_ctx_train(const struct llama_model * model), "use llama_model_n_ctx_train instead");563    DEPRECATED(LLAMA_API int32_t llama_n_embd     (const struct llama_model * model), "use llama_model_n_embd instead");564    DEPRECATED(LLAMA_API int32_t llama_n_layer    (const struct llama_model * model), "use llama_model_n_layer instead");565    DEPRECATED(LLAMA_API int32_t llama_n_head     (const struct llama_model * model), "use llama_model_n_head instead");566 567    DEPRECATED(LLAMA_API int32_t llama_n_vocab    (const struct llama_vocab * vocab), "use llama_vocab_n_tokens instead");568 569    LLAMA_API const struct llama_model * llama_get_model   (const struct llama_context * ctx);570    LLAMA_API           llama_memory_t   llama_get_memory  (const struct llama_context * ctx);571    LLAMA_API  enum llama_pooling_type   llama_pooling_type(const struct llama_context * ctx); // TODO: rename to llama_get_pooling_type572 573    LLAMA_API const struct llama_vocab * llama_model_get_vocab(const struct llama_model * model);574    LLAMA_API enum llama_rope_type       llama_model_rope_type(const struct llama_model * model);575 576    LLAMA_API int32_t llama_model_n_ctx_train  (const struct llama_model * model);577    LLAMA_API int32_t llama_model_n_embd       (const struct llama_model * model);578    LLAMA_API int32_t llama_model_n_embd_inp   (const struct llama_model * model);579    LLAMA_API int32_t llama_model_n_embd_out   (const struct llama_model * model);580    LLAMA_API int32_t llama_model_n_layer      (const struct llama_model * model);581    LLAMA_API int32_t llama_model_n_layer_nextn(const struct llama_model * model);582    LLAMA_API int32_t llama_model_n_head       (const struct llama_model * model);583    LLAMA_API int32_t llama_model_n_head_kv    (const struct llama_model * model);584    LLAMA_API int32_t llama_model_n_swa        (const struct llama_model * model);585 586    // Get the model's RoPE frequency scaling factor587    LLAMA_API float llama_model_rope_freq_scale_train(const struct llama_model * model);588 589    // Returns the number of classifier outputs (only valid for classifier models)590    // Undefined behavior for non-classifier models591    LLAMA_API uint32_t llama_model_n_cls_out(const struct llama_model * model);592 593    // Returns label of classifier output by index (<n_cls_out). Returns nullptr if no label provided594    LLAMA_API const char * llama_model_cls_label(const struct llama_model * model, uint32_t i);595 596    LLAMA_API enum llama_vocab_type llama_vocab_type(const struct llama_vocab * vocab);597 598    LLAMA_API int32_t llama_vocab_n_tokens(const struct llama_vocab * vocab);599 600    // Functions to access the model's GGUF metadata scalar values601    // - The functions return the length of the string on success, or -1 on failure602    // - The output string is always null-terminated and cleared on failure603    // - When retrieving a string, an extra byte must be allocated to account for the null terminator604    // - GGUF array values are not supported by these functions605 606    // Get metadata value as a string by key name607    LLAMA_API int32_t llama_model_meta_val_str(const struct llama_model * model, const char * key, char * buf, size_t buf_size);608 609    // Get the number of metadata key/value pairs610    LLAMA_API int32_t llama_model_meta_count(const struct llama_model * model);611 612    // Get sampling metadata key name. Returns nullptr if the key is invalid613    LLAMA_API const char * llama_model_meta_key_str(enum llama_model_meta_key key);614 615    // Get metadata key name by index616    LLAMA_API int32_t llama_model_meta_key_by_index(const struct llama_model * model, int32_t i, char * buf, size_t buf_size);617 618    // Get metadata value as a string by index619    LLAMA_API int32_t llama_model_meta_val_str_by_index(const struct llama_model * model, int32_t i, char * buf, size_t buf_size);620 621    // Get a string describing the model type622    LLAMA_API int32_t llama_model_desc(const struct llama_model * model, char * buf, size_t buf_size);623 624    // Get the model file type (quantization), e.g. LLAMA_FTYPE_MOSTLY_Q8_0625    LLAMA_API enum llama_ftype llama_model_ftype(const struct llama_model * model);626 627    // Returns the total size of all the tensors in the model in bytes628    LLAMA_API uint64_t llama_model_size(const struct llama_model * model);629 630    // Get the default chat template. Returns nullptr if not available631    // If name is NULL, returns the default chat template632    LLAMA_API const char * llama_model_chat_template(const struct llama_model * model, const char * name);633 634    // Returns the total number of parameters in the model635    LLAMA_API uint64_t llama_model_n_params(const struct llama_model * model);636 637    // Returns true if the model contains an encoder that requires llama_encode() call638    LLAMA_API bool llama_model_has_encoder(const struct llama_model * model);639 640    // Returns true if the model contains a decoder that requires llama_decode() call641    LLAMA_API bool llama_model_has_decoder(const struct llama_model * model);642 643    // For encoder-decoder models, this function returns id of the token that must be provided644    // to the decoder to start generating output sequence. For other models, it returns -1.645    LLAMA_API llama_token llama_model_decoder_start_token(const struct llama_model * model);646 647    // Returns true if the model is recurrent (like Mamba, RWKV, etc.)648    LLAMA_API bool llama_model_is_recurrent(const struct llama_model * model);649 650    // Returns true if the model is hybrid (like Jamba, Granite, etc.)651    LLAMA_API bool llama_model_is_hybrid(const struct llama_model * model);652 653    // Returns true if the model is diffusion-based (like LLaDA, Dream, etc.)654    LLAMA_API bool llama_model_is_diffusion(const struct llama_model * model);655 656    // Returns 0 on success657    LLAMA_API uint32_t llama_model_quantize(658            const char * fname_inp,659            const char * fname_out,660            const llama_model_quantize_params * params);661 662    //663    // Adapters664    //665 666    // Load a LoRA adapter from file667    // The adapter is valid as long as the associated model is not freed668    LLAMA_API struct llama_adapter_lora * llama_adapter_lora_init(669            struct llama_model * model,670            const char * path_lora);671 672    // Functions to access the adapter's GGUF metadata scalar values673    // - The functions return the length of the string on success, or -1 on failure674    // - The output string is always null-terminated and cleared on failure675    // - When retrieving a string, an extra byte must be allocated to account for the null terminator676    // - GGUF array values are not supported by these functions677 678    // Get metadata value as a string by key name679    LLAMA_API int32_t llama_adapter_meta_val_str(const struct llama_adapter_lora * adapter, const char * key, char * buf, size_t buf_size);680 681    // Get the number of metadata key/value pairs682    LLAMA_API int32_t llama_adapter_meta_count(const struct llama_adapter_lora * adapter);683 684    // Get metadata key name by index685    LLAMA_API int32_t llama_adapter_meta_key_by_index(const struct llama_adapter_lora * adapter, int32_t i, char * buf, size_t buf_size);686 687    // Get metadata value as a string by index688    LLAMA_API int32_t llama_adapter_meta_val_str_by_index(const struct llama_adapter_lora * adapter, int32_t i, char * buf, size_t buf_size);689 690    // Manually free a LoRA adapter691    // NOTE: loaded adapters that are not manually freed will be freed when the associated model is deleted692    LLAMA_API void llama_adapter_lora_free(struct llama_adapter_lora * adapter);693 694    // Get the invocation tokens if the current lora is an alora695    LLAMA_API uint64_t            llama_adapter_get_alora_n_invocation_tokens(const struct llama_adapter_lora * adapter);696    LLAMA_API const llama_token * llama_adapter_get_alora_invocation_tokens  (const struct llama_adapter_lora * adapter);697 698    // The following functions operate on a llama_context, hence the naming: llama_verb_...699 700    // Set LoRa adapters on the context. Will only modify if the adapters currently in context are different.701    LLAMA_API int32_t llama_set_adapters_lora(702            struct llama_context * ctx,703            struct llama_adapter_lora ** adapters,704            size_t n_adapters,705            float * scales);706 707    // Apply a loaded control vector to a llama_context, or if data is NULL, clear708    // the currently loaded vector.709    // n_embd should be the size of a single layer's control, and data should point710    // to an n_embd x n_layers buffer starting from layer 1.711    // il_start and il_end are the layer range the vector should apply to (both inclusive)712    // See llama_control_vector_load in common to load a control vector.713    LLAMA_API int32_t llama_set_adapter_cvec(714            struct llama_context * ctx,715                     const float * data,716                          size_t   len,717                         int32_t   n_embd,718                         int32_t   il_start,719                         int32_t   il_end);720 721    //722    // Memory723    //724 725    // Clear the memory contents726    // If data == true, the data buffers will also be cleared together with the metadata727    LLAMA_API void llama_memory_clear(728            llama_memory_t mem,729                      bool data);730 731    // Removes all tokens that belong to the specified sequence and have positions in [p0, p1)732    // Returns false if a partial sequence cannot be removed. Removing a whole sequence never fails733    // seq_id < 0 : match any sequence734    // p0 < 0     : [0,  p1]735    // p1 < 0     : [p0, inf)736    LLAMA_API bool llama_memory_seq_rm(737            llama_memory_t mem,738              llama_seq_id seq_id,739                 llama_pos p0,740                 llama_pos p1);741 742    // Copy all tokens that belong to the specified sequence to another sequence743    // p0 < 0 : [0,  p1]744    // p1 < 0 : [p0, inf)745    LLAMA_API void llama_memory_seq_cp(746            llama_memory_t mem,747              llama_seq_id seq_id_src,748              llama_seq_id seq_id_dst,749                 llama_pos p0,750                 llama_pos p1);751 752    // Removes all tokens that do not belong to the specified sequence753    LLAMA_API void llama_memory_seq_keep(754            llama_memory_t mem,755              llama_seq_id seq_id);756 757    // Adds relative position "delta" to all tokens that belong to the specified sequence and have positions in [p0, p1)758    // p0 < 0 : [0,  p1]759    // p1 < 0 : [p0, inf)760    LLAMA_API void llama_memory_seq_add(761            llama_memory_t mem,762              llama_seq_id seq_id,763                 llama_pos p0,764                 llama_pos p1,765                 llama_pos delta);766 767    // Integer division of the positions by factor of `d > 1`768    // p0 < 0 : [0,  p1]769    // p1 < 0 : [p0, inf)770    LLAMA_API void llama_memory_seq_div(771            llama_memory_t mem,772              llama_seq_id seq_id,773                 llama_pos p0,774                 llama_pos p1,775                       int d);776 777    // Returns the smallest position present in the memory for the specified sequence778    // This is typically non-zero only for SWA caches779    // Note that all positions in the range [pos_min, pos_max] are guaranteed to be present in the memory780    // Return -1 if the sequence is empty781    LLAMA_API llama_pos llama_memory_seq_pos_min(782            llama_memory_t mem,783              llama_seq_id seq_id);784 785    // Returns the largest position present in the memory for the specified sequence786    // Note that all positions in the range [pos_min, pos_max] are guaranteed to be present in the memory787    // Return -1 if the sequence is empty788    LLAMA_API llama_pos llama_memory_seq_pos_max(789            llama_memory_t mem,790              llama_seq_id seq_id);791 792    // Check if the memory supports shifting793    LLAMA_API bool llama_memory_can_shift(llama_memory_t mem);794 795    //796    // State / sessions797    //798 799    // Returns the *actual* size in bytes of the state800    // (logits, embedding and memory)801    // Only use when saving the state, not when restoring it, otherwise the size may be too small.802    LLAMA_API size_t llama_state_get_size(struct llama_context * ctx);803    LLAMA_API DEPRECATED(size_t llama_get_state_size(struct llama_context * ctx),804        "use llama_state_get_size instead");805 806    // Copies the state to the specified destination address.807    // Destination needs to have allocated enough memory.808    // Returns the number of bytes copied809    LLAMA_API size_t llama_state_get_data(810            struct llama_context * ctx,811                         uint8_t * dst,812                          size_t   size);813    LLAMA_API DEPRECATED(size_t llama_copy_state_data(814            struct llama_context * ctx,815                         uint8_t * dst),816        "use llama_state_get_data instead");817 818    // Set the state reading from the specified address819    // Returns the number of bytes read820    LLAMA_API size_t llama_state_set_data(821            struct llama_context * ctx,822                   const uint8_t * src,823                          size_t   size);824    LLAMA_API DEPRECATED(size_t llama_set_state_data(825            struct llama_context * ctx,826                   const uint8_t * src),827        "use llama_state_set_data instead");828 829    // Save/load session file830    LLAMA_API bool llama_state_load_file(831            struct llama_context * ctx,832                      const char * path_session,833                     llama_token * tokens_out,834                          size_t   n_token_capacity,835                          size_t * n_token_count_out);836    LLAMA_API DEPRECATED(bool llama_load_session_file(837            struct llama_context * ctx,838                      const char * path_session,839                     llama_token * tokens_out,840                          size_t   n_token_capacity,841                          size_t * n_token_count_out),842        "use llama_state_load_file instead");843 844    LLAMA_API bool llama_state_save_file(845            struct llama_context * ctx,846                      const char * path_session,847               const llama_token * tokens,848                          size_t   n_token_count);849    LLAMA_API DEPRECATED(bool llama_save_session_file(850            struct llama_context * ctx,851                      const char * path_session,852               const llama_token * tokens,853                          size_t   n_token_count),854        "use llama_state_save_file instead");855 856    // Get the exact size needed to copy the state of a single sequence857    LLAMA_API size_t llama_state_seq_get_size(858            struct llama_context * ctx,859                    llama_seq_id   seq_id);860 861    // Copy the state of a single sequence into the specified buffer862    LLAMA_API size_t llama_state_seq_get_data(863            struct llama_context * ctx,864                         uint8_t * dst,865                          size_t   size,866                    llama_seq_id   seq_id);867 868    // Copy the sequence data (originally copied with `llama_state_seq_get_data`) into the specified sequence869    // Returns:870    //  - Positive: Ok871    //  - Zero: Failed to load872    LLAMA_API size_t llama_state_seq_set_data(873            struct llama_context * ctx,874                   const uint8_t * src,875                          size_t   size,876                    llama_seq_id   dest_seq_id);877 878    LLAMA_API size_t llama_state_seq_save_file(879            struct llama_context * ctx,880                      const char * filepath,881                    llama_seq_id   seq_id,882               const llama_token * tokens,883                          size_t   n_token_count);884 885    LLAMA_API size_t llama_state_seq_load_file(886            struct llama_context * ctx,887                      const char * filepath,888                    llama_seq_id   dest_seq_id,889                     llama_token * tokens_out,890                          size_t   n_token_capacity,891                          size_t * n_token_count_out);892 893#define LLAMA_STATE_SEQ_FLAGS_NONE 0894 895// for backwards-compat896#define LLAMA_STATE_SEQ_FLAGS_SWA_ONLY 1897 898// work only with partial states, such as SWA KV cache or recurrent cache (e.g. Mamba)899#define LLAMA_STATE_SEQ_FLAGS_PARTIAL_ONLY 1900 901// Keeps the tensor data on device buffers (i.e. not accessible in host memory, but faster save/load).902// Getting the state for a seq_id with this flag invalidates all prior states gotten for that seq_id with this flag.903#define LLAMA_STATE_SEQ_FLAGS_ON_DEVICE 2904 905    typedef uint32_t llama_state_seq_flags;906 907    LLAMA_API size_t llama_state_seq_get_size_ext(908            struct llama_context * ctx,909                    llama_seq_id   seq_id,910           llama_state_seq_flags   flags);911 912    LLAMA_API size_t llama_state_seq_get_data_ext(913            struct llama_context * ctx,914                         uint8_t * dst,915                          size_t   size,916                    llama_seq_id   seq_id,917           llama_state_seq_flags   flags);918 919    LLAMA_API size_t llama_state_seq_set_data_ext(920            struct llama_context * ctx,921                   const uint8_t * src,922                          size_t   size,923                    llama_seq_id   dest_seq_id,924           llama_state_seq_flags   flags);925 926    //927    // Decoding928    //929 930    // Return batch for single sequence of tokens931    // The sequence ID will be fixed to 0932    // The position of the tokens will be tracked automatically by llama_decode933    //934    // NOTE: this is a helper function to facilitate transition to the new batch API - avoid using it935    //936    LLAMA_API struct llama_batch llama_batch_get_one(937                  llama_token * tokens,938                      int32_t   n_tokens);939 940    // Allocates a batch of tokens on the heap that can hold a maximum of n_tokens941    // Each token can be assigned up to n_seq_max sequence ids942    // The batch has to be freed with llama_batch_free()943    // If embd != 0, llama_batch.embd will be allocated with size of n_tokens * embd * sizeof(float)944    // Otherwise, llama_batch.token will be allocated to store n_tokens llama_token945    // The rest of the llama_batch members are allocated with size n_tokens946    // All members are left uninitialized947    LLAMA_API struct llama_batch llama_batch_init(948            int32_t n_tokens,949            int32_t embd,950            int32_t n_seq_max);951 952    // Frees a batch of tokens allocated with llama_batch_init()953    LLAMA_API void llama_batch_free(struct llama_batch batch);954 955    // Process a batch of tokens.956    // In contrast to llama_decode() - this call does not use KV cache.957    // For encode-decoder contexts, processes the batch using the encoder.958    // Can store the encoder output internally for later use by the decoder's cross-attention layers.959    //   0 - success960    // < 0 - error. the memory state is restored to the state before this call961    LLAMA_API int32_t llama_encode(962            struct llama_context * ctx,963              struct llama_batch   batch);964 965    // Process a batch of tokens.966    // Requires the context to have a memory.967    // For encode-decoder contexts, processes the batch using the decoder.968    // Positive return values does not mean a fatal error, but rather a warning.969    // Upon fatal-error or abort, the ubatches that managed to be been processed will remain in the memory state of the context970    //   To handle this correctly, query the memory state using llama_memory_seq_pos_min() and llama_memory_seq_pos_max()971    // Upon other return values, the memory state is restored to the state before this call972    //    0 - success973    //    1 - could not find a KV slot for the batch (try reducing the size of the batch or increase the context)974    //    2 - aborted     (processed ubatches will remain in the context's memory)975    //   -1 - invalid input batch976    // < -1 - fatal error (processed ubatches will remain in the context's memory)977    LLAMA_API int32_t llama_decode(978            struct llama_context * ctx,979              struct llama_batch   batch);980 981    // Set the number of threads used for decoding982    // n_threads is the number of threads used for generation (single token)983    // n_threads_batch is the number of threads used for prompt and batch processing (multiple tokens)984    LLAMA_API void llama_set_n_threads(struct llama_context * ctx, int32_t n_threads, int32_t n_threads_batch);985 986    // Get the number of threads used for generation of a single token.987    LLAMA_API int32_t llama_n_threads(struct llama_context * ctx);988 989    // Get the number of threads used for prompt and batch processing (multiple token).990    LLAMA_API int32_t llama_n_threads_batch(struct llama_context * ctx);991 992    // Set whether the context outputs embeddings or not993    // TODO: rename to avoid confusion with llama_get_embeddings()994    LLAMA_API void llama_set_embeddings(struct llama_context * ctx, bool embeddings);995 996    // Set whether to use causal attention or not997    // If set to true, the model will only attend to the past tokens998    LLAMA_API void llama_set_causal_attn(struct llama_context * ctx, bool causal_attn);999 1000    // Set whether the model is in warmup mode or not1001    // If true, all model tensors are activated during llama_decode() to load and cache their weights.1002    //1003    // note: using this can cause extra graph reallocations because it changes the graph topology with MoE models,1004    //       so it is generally not recommended to use in practice. will be removed in the future1005    DEPRECATED(LLAMA_API void llama_set_warmup(struct llama_context * ctx, bool warmup),1006            "user code should do warmup runs manually [TAG_LLAMA_GRAPH_NO_WARMUP]");1007 1008    // Set abort callback1009    LLAMA_API void llama_set_abort_callback(struct llama_context * ctx, ggml_abort_callback abort_callback, void * abort_callback_data);1010 1011    // Wait until all computations are finished1012    // This is automatically done when using one of the functions below to obtain the computation results1013    // and is not necessary to call it explicitly in most cases1014    LLAMA_API void llama_synchronize(struct llama_context * ctx);1015 1016    // Token logits obtained from the last call to llama_decode()1017    // The logits for which llama_batch.logits[i] != 0 are stored contiguously1018    // in the order they have appeared in the batch.1019    // Rows: number of tokens for which llama_batch.logits[i] != 01020    // Cols: n_vocab1021    // TODO: deprecate in favor of llama_get_logits_ith() (ref: https://github.com/ggml-org/llama.cpp/pull/14853#issuecomment-3113143522)1022    LLAMA_API float * llama_get_logits(struct llama_context * ctx);1023 1024    // Logits for the ith token. For positive indices, Equivalent to:1025    // llama_get_logits(ctx) + ctx->output_ids[i]*n_vocab1026    // Negative indices can be used to access logits in reverse order, -1 is the last logit.1027    // returns NULL for invalid ids.1028    LLAMA_API float * llama_get_logits_ith(struct llama_context * ctx, int32_t i);1029 1030    // Get all output token embeddings.1031    // when pooling_type == LLAMA_POOLING_TYPE_NONE or when using a generative model,1032    // the embeddings for which llama_batch.logits[i] != 0 are stored contiguously1033    // in the order they have appeared in the batch.1034    // shape: [n_outputs*n_embd]1035    // Otherwise, returns NULL.1036    // TODO: deprecate in favor of llama_get_embeddings_ith() (ref: https://github.com/ggml-org/llama.cpp/pull/14853#issuecomment-3113143522)1037    LLAMA_API float * llama_get_embeddings(struct llama_context * ctx);1038 1039    // Get the embeddings for the ith token. For positive indices, Equivalent to:1040    // llama_get_embeddings(ctx) + ctx->output_ids[i]*n_embd1041    // Negative indices can be used to access embeddings in reverse order, -1 is the last embedding.1042    // shape: [n_embd] (1-dimensional)1043    // returns NULL for invalid ids.1044    LLAMA_API float * llama_get_embeddings_ith(struct llama_context * ctx, int32_t i);1045 1046    // Get the embeddings for a sequence id1047    // Returns NULL if pooling_type is LLAMA_POOLING_TYPE_NONE1048    // when pooling_type == LLAMA_POOLING_TYPE_RANK, returns float[n_cls_out] with the rank(s) of the sequence1049    // otherwise: float[n_embd] (1-dimensional)1050    LLAMA_API float * llama_get_embeddings_seq(struct llama_context * ctx, llama_seq_id seq_id);1051 1052    //1053    // backend sampling API [EXPERIMENTAL]1054    // note: use only if the llama_context was created with at least one llama_sampler_seq_config1055    //1056 1057    // Get the backend sampled token for the ith token.1058    // With multiple outputs, sampler state advances when the token is accepted,1059    // not when it is read through this function.1060    // When accepting multiple outputs, accept a contiguous prefix in output order.1061    // Returns LLAMA_TOKEN_NULL if no token was sampled.1062    LLAMA_API llama_token llama_get_sampled_token_ith(struct llama_context * ctx, int32_t i);1063 1064    // Get the backend sampled probabilities for the ith token1065    // The index matches llama_get_sampled_token_ith().1066    // Returns NULL if no probabilities were generated.1067    LLAMA_API float *  llama_get_sampled_probs_ith      (struct llama_context * ctx, int32_t i);1068    LLAMA_API uint32_t llama_get_sampled_probs_count_ith(struct llama_context * ctx, int32_t i);1069 1070    // Get the backend sampled logits for the ith token1071    // Returns NULL if no logits were sampled.1072    LLAMA_API float *  llama_get_sampled_logits_ith      (struct llama_context * ctx, int32_t i);1073    LLAMA_API uint32_t llama_get_sampled_logits_count_ith(struct llama_context * ctx, int32_t i);1074 1075    // Get the backend sampled candidates (token ids) for the ith token1076    // These are needed to map probability/logit indices to vocab token ids.1077    // Returns NULL if no candidates were sampled.1078    LLAMA_API llama_token * llama_get_sampled_candidates_ith      (struct llama_context * ctx, int32_t i);1079    LLAMA_API uint32_t      llama_get_sampled_candidates_count_ith(struct llama_context * ctx, int32_t i);1080 1081    //1082    // Vocab1083    //1084 1085    LLAMA_API const char * llama_vocab_get_text(const struct llama_vocab * vocab, llama_token token);1086 1087    LLAMA_API float llama_vocab_get_score(const struct llama_vocab * vocab, llama_token token);1088 1089    LLAMA_API enum llama_token_attr llama_vocab_get_attr(const struct llama_vocab * vocab, llama_token token);1090 1091    // Check if the token is supposed to end generation (end-of-generation, eg. EOS, EOT, etc.)1092    LLAMA_API bool llama_vocab_is_eog(const struct llama_vocab * vocab, llama_token token);1093 1094    // Identify if Token Id is a control token or a render-able token1095    LLAMA_API bool llama_vocab_is_control(const struct llama_vocab * vocab, llama_token token);1096 1097    // Special tokens1098    LLAMA_API llama_token llama_vocab_bos(const struct llama_vocab * vocab); // beginning-of-sentence1099    LLAMA_API llama_token llama_vocab_eos(const struct llama_vocab * vocab); // end-of-sentence1100    LLAMA_API llama_token llama_vocab_eot(const struct llama_vocab * vocab); // end-of-turn1101    LLAMA_API llama_token llama_vocab_sep(const struct llama_vocab * vocab); // sentence separator1102    LLAMA_API llama_token llama_vocab_nl (const struct llama_vocab * vocab); // next-line1103    LLAMA_API llama_token llama_vocab_pad(const struct llama_vocab * vocab); // padding1104    LLAMA_API llama_token llama_vocab_mask(const struct llama_vocab * vocab); // mask1105 1106    LLAMA_API bool llama_vocab_get_add_bos(const struct llama_vocab * vocab);1107    LLAMA_API bool llama_vocab_get_add_eos(const struct llama_vocab * vocab);1108    LLAMA_API bool llama_vocab_get_add_sep(const struct llama_vocab * vocab);1109 1110    // model-specific suppress tokens (gguf key: tokenizer.ggml.suppress_tokens)1111    LLAMA_API const llama_token * llama_vocab_get_suppress_tokens(const struct llama_vocab * vocab, int32_t * n_suppress_tokens);1112 1113    LLAMA_API llama_token llama_vocab_fim_pre(const struct llama_vocab * vocab);1114    LLAMA_API llama_token llama_vocab_fim_suf(const struct llama_vocab * vocab);1115    LLAMA_API llama_token llama_vocab_fim_mid(const struct llama_vocab * vocab);1116    LLAMA_API llama_token llama_vocab_fim_pad(const struct llama_vocab * vocab);1117    LLAMA_API llama_token llama_vocab_fim_rep(const struct llama_vocab * vocab);1118    LLAMA_API llama_token llama_vocab_fim_sep(const struct llama_vocab * vocab);1119 1120    DEPRECATED(LLAMA_API const char * llama_token_get_text(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_get_text instead");1121    DEPRECATED(LLAMA_API float llama_token_get_score(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_get_score instead");1122    DEPRECATED(LLAMA_API enum llama_token_attr llama_token_get_attr(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_get_attr instead");1123    DEPRECATED(LLAMA_API bool llama_token_is_eog(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_is_eog instead");1124    DEPRECATED(LLAMA_API bool llama_token_is_control(const struct llama_vocab * vocab, llama_token token), "use llama_vocab_is_control instead");1125    DEPRECATED(LLAMA_API llama_token llama_token_bos(const struct llama_vocab * vocab), "use llama_vocab_bos instead");1126    DEPRECATED(LLAMA_API llama_token llama_token_eos(const struct llama_vocab * vocab), "use llama_vocab_eos instead");1127    DEPRECATED(LLAMA_API llama_token llama_token_eot(const struct llama_vocab * vocab), "use llama_vocab_eot instead");1128    DEPRECATED(LLAMA_API llama_token llama_token_cls(const struct llama_vocab * vocab), "use llama_vocab_cls instead");1129    DEPRECATED(LLAMA_API llama_token llama_token_sep(const struct llama_vocab * vocab), "use llama_vocab_sep instead");1130    DEPRECATED(LLAMA_API llama_token llama_token_nl (const struct llama_vocab * vocab), "use llama_vocab_nl instead");1131    DEPRECATED(LLAMA_API llama_token llama_token_pad(const struct llama_vocab * vocab), "use llama_vocab_pad instead");1132    DEPRECATED(LLAMA_API bool llama_add_bos_token(const struct llama_vocab * vocab), "use llama_vocab_get_add_bos instead");1133    DEPRECATED(LLAMA_API bool llama_add_eos_token(const struct llama_vocab * vocab), "use llama_vocab_get_add_eos instead");1134    DEPRECATED(LLAMA_API llama_token llama_token_fim_pre(const struct llama_vocab * vocab), "use llama_vocab_fim_pre instead");1135    DEPRECATED(LLAMA_API llama_token llama_token_fim_suf(const struct llama_vocab * vocab), "use llama_vocab_fim_suf instead");1136    DEPRECATED(LLAMA_API llama_token llama_token_fim_mid(const struct llama_vocab * vocab), "use llama_vocab_fim_mid instead");1137    DEPRECATED(LLAMA_API llama_token llama_token_fim_pad(const struct llama_vocab * vocab), "use llama_vocab_fim_pad instead");1138    DEPRECATED(LLAMA_API llama_token llama_token_fim_rep(const struct llama_vocab * vocab), "use llama_vocab_fim_rep instead");1139    DEPRECATED(LLAMA_API llama_token llama_token_fim_sep(const struct llama_vocab * vocab), "use llama_vocab_fim_sep instead");1140 1141    // CLS is equivalent to BOS1142    DEPRECATED(LLAMA_API llama_token llama_vocab_cls(const struct llama_vocab * vocab), // classification1143            "use llama_vocab_bos instead");1144 1145    //1146    // Tokenization1147    //1148    // The API is thread-safe.1149    //1150 1151    /// @details Convert the provided text into tokens.1152    /// @param tokens The tokens pointer must be large enough to hold the resulting tokens.1153    /// @return Returns the number of tokens on success, no more than n_tokens_max1154    /// @return Returns a negative number on failure - the number of tokens that would have been returned1155    /// @return Returns INT32_MIN on overflow (e.g., tokenization result size exceeds int32_t limit)1156    /// @param add_special Allow to add BOS and EOS tokens if model is configured to do so.1157    /// @param parse_special Allow tokenizing special and/or control tokens which otherwise are not exposed and treated1158    ///                      as plaintext. Does not insert a leading space.1159    LLAMA_API int32_t llama_tokenize(1160        const struct llama_vocab * vocab,1161                      const char * text,1162                         int32_t   text_len,1163                     llama_token * tokens,1164                         int32_t   n_tokens_max,1165                            bool   add_special,1166                            bool   parse_special);1167 1168    // Token Id -> Piece.1169    // Uses the vocabulary in the provided context.1170    // Does not write null terminator to the buffer.1171    // User can skip up to 'lstrip' leading spaces before copying (useful when encoding/decoding multiple tokens with 'add_space_prefix')1172    // @param special If true, special tokens are rendered in the output.1173    LLAMA_API int32_t llama_token_to_piece(1174              const struct llama_vocab * vocab,1175                           llama_token   token,1176                                  char * buf,1177                               int32_t   length,1178                               int32_t   lstrip,1179                                  bool   special);1180 1181    /// @details Convert the provided tokens into text (inverse of llama_tokenize()).1182    /// @param text The char pointer must be large enough to hold the resulting text.1183    /// @return Returns the number of chars/bytes on success, no more than text_len_max.1184    /// @return Returns a negative number on failure - the number of chars/bytes that would have been returned.1185    /// @param remove_special Allow to remove BOS and EOS tokens if model is configured to do so.1186    /// @param unparse_special If true, special tokens are rendered in the output.1187    LLAMA_API int32_t llama_detokenize(1188        const struct llama_vocab * vocab,1189               const llama_token * tokens,1190                         int32_t   n_tokens,1191                            char * text,1192                         int32_t   text_len_max,1193                            bool   remove_special,1194                            bool   unparse_special);1195 1196    //1197    // Chat templates1198    //1199 1200    /// Apply chat template. Inspired by hf apply_chat_template() on python.

Showing the first 1,200 of 1626 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai