Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
llama-context.h395 linesDownload Raw Back to src
1#pragma once2 3#include "llama.h"4#include "llama-ext.h"5#include "llama-cparams.h"6#include "llama-graph.h"7#include "llama-adapter.h"8#include "llama-impl.h"9#include "llama-memory.h"10 11#include "ggml-cpp.h"12#include "ggml-opt.h"13 14#include <map>15#include <vector>16 17struct llama_model;18class llama_batch_allocr;19 20class llama_io_read_i;21class llama_io_write_i;22 23// "memory" as in abstract memory for the context24struct llama_memory_i;25struct llama_memory_context_i;26 27// stores copy of the memory in device buffer. used for fast state save/load28struct llama_memory_buffer {29    int n_tensors = 0;30    size_t total_size = 0;31 32    ggml_backend_buffer_ptr buf;33 34    ggml_context_ptr ctx;35 36    std::vector<ggml_tensor *> org;37    std::vector<ggml_tensor *> cpy;38};39 40using llama_memory_buffers = std::map<ggml_backend_buffer_type_t, llama_memory_buffer>;41 42struct llama_context {43    // init scheduler and compute buffers, reserve worst-case graphs44    llama_context(45            const llama_model & model,46                  llama_context_params params);47 48    ~llama_context();49 50    // reserve a new backend scheduler (if needed)51    // for example, when:52    //   - changing loras53    //   - changing samplers54    //   - changing attention type55    //   - etc.56    void sched_reserve();57 58    void synchronize();59 60    const llama_model   & get_model()   const;61    const llama_cparams & get_cparams() const;62 63    ggml_backend_sched_t get_sched() const;64 65    uint32_t n_ctx()     const;66    uint32_t n_ctx_seq() const;67    uint32_t n_batch()   const;68    uint32_t n_ubatch()  const;69    uint32_t n_seq_max() const;70 71    uint32_t n_threads()       const;72    uint32_t n_threads_batch() const;73 74    llama_memory_t get_memory() const;75 76    // return true if the memory was updated77    bool memory_update(bool optimize);78 79    enum llama_pooling_type pooling_type() const;80 81    float * get_logits();82    float * get_logits_ith(int32_t i);83 84    float * get_embeddings();85    float * get_embeddings_ith(int32_t i);86    float * get_embeddings_seq(llama_seq_id seq_id);87 88    float * get_embeddings_nextn();89    float * get_embeddings_nextn_ith(int32_t i);90 91    float * get_embeddings_layer_inp(uint32_t lid);92 93    llama_token * get_sampled_tokens() const;94    llama_token   get_sampled_token_ith(int32_t idx);95 96    float * get_sampled_logits_ith(int32_t idx);97    size_t  get_sampled_logits_count(int32_t idx);98 99    float * get_sampled_probs_ith(int32_t idx);100    size_t  get_sampled_probs_count(int32_t idx);101 102    const llama_token * get_sampled_candidates_ith(int32_t idx);103    size_t get_sampled_candidates_count(int32_t idx);104 105    void attach_threadpool(106            ggml_threadpool_t threadpool,107            ggml_threadpool_t threadpool_batch);108 109    void detach_threadpool();110 111    void set_n_threads(int32_t n_threads, int32_t n_threads_batch);112 113    void set_abort_callback(bool (*abort_callback)(void * data), void * abort_callback_data);114 115    void set_embeddings (bool value);116    void set_embeddings_nextn(bool value, bool masked);117    void set_embeddings_layer_inp(uint32_t lid, bool enable);118    void set_nextn_layer_offset(int32_t offset);119    void set_causal_attn(bool value);120    void set_warmup(bool value);121 122    void set_adapters_lora(llama_adapter_lora ** adapters, size_t n_adapters, float * scales);123 124    bool adapters_lora_are_same(llama_adapter_lora ** adapters, size_t n_adapters, float * scales);125 126    bool set_adapter_cvec(127            const float * data,128                 size_t   len,129                int32_t   n_embd,130                int32_t   il_start,131                int32_t   il_end);132 133    // process a single ubatch with a specific graph type134    // if memory_context is provided, it will be applied first to the context's memory135    // ret contains the status of the graph computation136    // returns nullptr only if ret != GGML_STATUS_SUCCESS137    llm_graph_result * process_ubatch(138                const llama_ubatch & ubatch,139                    llm_graph_type   gtype,140            llama_memory_context_i * mctx,141                       ggml_status & ret);142 143    int encode(const llama_batch & batch_inp);144    int decode(const llama_batch & batch_inp);145 146    //147    // state save/load148    //149 150    size_t state_get_size();151    size_t state_get_data(      uint8_t * dst, size_t size);152    size_t state_set_data(const uint8_t * src, size_t size);153 154    size_t state_seq_get_size(llama_seq_id seq_id, llama_state_seq_flags flags);155 156    size_t state_seq_get_data(llama_seq_id seq_id,       uint8_t * dst, size_t size, llama_state_seq_flags flags);157    size_t state_seq_set_data(llama_seq_id seq_id, const uint8_t * src, size_t size, llama_state_seq_flags flags);158 159    bool state_load_file(160            const char * filepath,161           llama_token * tokens_out,162                size_t   n_token_capacity,163                size_t * n_token_count_out);164 165    bool state_save_file(166            const char * filepath,167     const llama_token * tokens,168                size_t   n_token_count);169 170    size_t state_seq_load_file(171          llama_seq_id   seq_id,172            const char * filepath,173           llama_token * tokens_out,174                size_t   n_token_capacity,175                size_t * n_token_count_out);176 177    size_t state_seq_save_file(178          llama_seq_id   seq_id,179            const char * filepath,180     const llama_token * tokens,181                size_t   n_token_count);182 183    //184    // perf185    //186 187    llama_perf_context_data perf_get_data() const;188    void perf_reset();189 190    llama_memory_breakdown memory_breakdown() const;191 192    //193    // training194    //195 196    void opt_init(struct llama_model * model, struct llama_opt_params lopt_params);197 198    // TODO: more flexible combinations of logical/physical batch size and context size199    void opt_epoch(200            ggml_opt_dataset_t      dataset,201            ggml_opt_result_t       result_train,202            ggml_opt_result_t       result_eval,203            int64_t                 idata_split,204            ggml_opt_epoch_callback callback_train,205            ggml_opt_epoch_callback callback_eval);206 207    void opt_epoch_iter(208            ggml_opt_dataset_t               dataset,209            ggml_opt_result_t                result,210            const std::vector<llama_token> & tokens,211            const std::vector<llama_token> & labels_sparse,212            llama_batch                    & batch,213            ggml_opt_epoch_callback          callback,214            bool                             train,215            int64_t                          idata_in_loop,216            int64_t                          ndata_in_loop,217            int64_t                          t_loop_start);218 219private:220    //221    // output222    //223 224    // Make sure enough space is available for outputs.225    // Returns max number of outputs for which space was reserved.226    uint32_t output_reserve(int32_t n_outputs);227 228    void output_reorder();229 230    // map the output row index `i` to batch index231    int64_t output_resolve_row(int32_t i) const;232 233    // async-copy enabled layer-input tensors (per cparams.output_layer_inp)234    // from backend into host-side embd_layer_inp buffers235    void extract_layer_inputs(const llm_graph_result * res, size_t token_offset, size_t n_tokens);236 237    //238    // graph239    //240 241public:242    uint32_t graph_max_nodes(uint32_t n_tokens) const;243 244    // can reuse the llm_graph_result instance of the context (for example to update a memory module)245    llm_graph_result * get_gf_res_reserve() const;246 247    // returns the result of ggml_backend_sched_graph_compute_async execution248    ggml_status graph_compute(ggml_cgraph * gf, bool batched);249 250    // reserve a graph with a dummy ubatch of the specified size251    ggml_cgraph * graph_reserve(252        uint32_t n_tokens, uint32_t n_seqs, uint32_t n_outputs, const llama_memory_context_i * mctx, bool split_only = false, size_t * sizes = nullptr);253 254    bool set_sampler(llama_seq_id seq_id, llama_sampler * sampler);255 256private:257    llm_graph_params graph_params(258                        llm_graph_result * res,259                      const llama_ubatch & ubatch,260            const llama_memory_context_i * mctx,261                          llm_graph_type   gtype) const;262 263    llm_graph_cb graph_get_cb() const;264 265    // disable auto fused ops (Flash Attention, Gated Delta Net) whose op lands on a device266    // that differs from the layer it belongs to (usually due to missing backend support)267    void resolve_fused_ops(const llama_memory_context_i * mctx, uint32_t n_seqs);268 269    // TODO: read/write lora adapters and cvec270    size_t state_write_data(llama_io_write_i & io);271    size_t state_read_data (llama_io_read_i  & io);272 273    size_t state_seq_write_data(llama_io_write_i & io, llama_seq_id seq_id, llama_state_seq_flags flags);274    size_t state_seq_read_data (llama_io_read_i  & io, llama_seq_id seq_id, llama_state_seq_flags flags);275 276    //277    // members278    //279 280    const llama_model & model;281 282    llama_cparams cparams;283 284    llama_adapter_cvec_ptr  cvec;285    llama_adapter_loras_ptr loras;286 287    llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably288 289    llama_memory_ptr memory;290 291    // decode output (2-dimensional array: [n_outputs][n_vocab])292    buffer_view<float> logits = {nullptr, 0};293 294    // embeddings output (2-dimensional array: [n_outputs][n_embd])295    // populated only when pooling_type == LLAMA_POOLING_TYPE_NONE296    buffer_view<float> embd = {nullptr, 0};297 298    // hidden state required by the nextn layers (2-dimensional array: [n_outputs][n_embd])299    // populated only when cparams.embeddings_nextn is enabled and the model graph300    // sets llm_graph_result::t_h_nextn301    buffer_view<float> embd_nextn = {nullptr, 0};302 303    // host buffers for output layer input embeddings, per layer304    // populated when cparams.output_layer_inp[il] is true305    std::vector<buffer_view<float>> embd_layer_inp;306 307    struct sampling_info {308        // !samplers.empty() to check if any samplers are active309        std::map<llama_seq_id, llama_sampler *> samplers;310 311        buffer_view<float>       logits     = {nullptr, 0};312        buffer_view<llama_token> sampled    = {nullptr, 0};313        buffer_view<float>       probs      = {nullptr, 0};314        buffer_view<llama_token> candidates = {nullptr, 0};315 316        std::vector<uint32_t> logits_count;317        std::vector<uint32_t> probs_count;318        std::vector<uint32_t> candidates_count;319 320        // optimization321        std::vector<llama_token> token_ids_full_vocab;322    };323 324    sampling_info sampling;325 326    // sequence embeddings output (map of [n_embd] vectors)327    // populated only when pooling_type != LLAMA_POOLING_TYPE_NONE328    std::map<llama_seq_id, std::vector<float>> embd_seq;329 330    // reuse the batch_allocr to avoid unnecessary memory allocations331    std::unique_ptr<llama_batch_allocr> balloc;332 333    uint32_t n_outputs = 0; // number of actually-used outputs in the current ubatch or last logical batch334 335    std::vector<int32_t> output_ids; // map batch token positions to ids of the logits and embd buffers336 337    struct swap_info {338        uint32_t i0;339        uint32_t i1;340    };341 342    std::vector<swap_info> output_swaps;343 344    ggml_backend_sched_ptr sched;345 346    bool sched_need_reserve = true;347 348    ggml_backend_t backend_cpu = nullptr;349    std::vector<ggml_backend_ptr> backends;350 351    // training352    ggml_opt_context_t opt_ctx = nullptr;353 354    ggml_threadpool_t threadpool       = nullptr;355    ggml_threadpool_t threadpool_batch = nullptr;356 357    ggml_abort_callback abort_callback      = nullptr;358    void *              abort_callback_data = nullptr;359 360    std::vector<std::pair<ggml_backend_t, ggml_backend_set_n_threads_t>> set_n_threads_fns;361 362    // pointers and buffer types used for the compute buffer of each backend363    std::vector<ggml_backend_t>             backend_ptrs;364    std::vector<ggml_backend_buffer_type_t> backend_buft;365    std::vector<size_t>                     backend_buf_exp_size; // expected buffer sizes366 367    llm_graph_result_ptr gf_res_prev;368    llm_graph_result_ptr gf_res_reserve;369 370    // host buffer for the model output (logits and embeddings)371    ggml_backend_buffer_ptr buf_output;372 373    // keep copies of the per-sequence memory on the device374    std::map<llama_seq_id, llama_memory_buffers> mem_storage;375 376    bool has_evaluated_once = false;377 378    // env: LLAMA_GRAPH_REUSE_DISABLE379    bool graph_reuse_disable = false;380 381    // perf382    mutable int64_t t_start_us  = 0;383    mutable int64_t t_load_us   = 0;384    mutable int64_t t_p_eval_us = 0;385    mutable int64_t t_eval_us   = 0;386 387    mutable int64_t t_compute_start_us = 0;388    mutable int64_t n_queued_tokens    = 0;389 390    mutable int32_t n_p_eval = 0; // number of tokens in eval calls for the prompt (with batch size > 1)391    mutable int32_t n_eval   = 0; // number of eval calls392 393    mutable int32_t n_reused = 0; // number of times the previous graph was reused394};395 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai