Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#pragma once2 3#include "llama.h"4#include "llama-ext.h"5#include "llama-cparams.h"6#include "llama-graph.h"7#include "llama-adapter.h"8#include "llama-impl.h"9#include "llama-memory.h"10 11#include "ggml-cpp.h"12#include "ggml-opt.h"13 14#include <map>15#include <vector>16 17struct llama_model;18class llama_batch_allocr;19 20class llama_io_read_i;21class llama_io_write_i;22 23// "memory" as in abstract memory for the context24struct llama_memory_i;25struct llama_memory_context_i;26 27// stores copy of the memory in device buffer. used for fast state save/load28struct llama_memory_buffer {29 int n_tensors = 0;30 size_t total_size = 0;31 32 ggml_backend_buffer_ptr buf;33 34 ggml_context_ptr ctx;35 36 std::vector<ggml_tensor *> org;37 std::vector<ggml_tensor *> cpy;38};39 40using llama_memory_buffers = std::map<ggml_backend_buffer_type_t, llama_memory_buffer>;41 42struct llama_context {43 // init scheduler and compute buffers, reserve worst-case graphs44 llama_context(45 const llama_model & model,46 llama_context_params params);47 48 ~llama_context();49 50 // reserve a new backend scheduler (if needed)51 // for example, when:52 // - changing loras53 // - changing samplers54 // - changing attention type55 // - etc.56 void sched_reserve();57 58 void synchronize();59 60 const llama_model & get_model() const;61 const llama_cparams & get_cparams() const;62 63 ggml_backend_sched_t get_sched() const;64 65 uint32_t n_ctx() const;66 uint32_t n_ctx_seq() const;67 uint32_t n_batch() const;68 uint32_t n_ubatch() const;69 uint32_t n_seq_max() const;70 71 uint32_t n_threads() const;72 uint32_t n_threads_batch() const;73 74 llama_memory_t get_memory() const;75 76 // return true if the memory was updated77 bool memory_update(bool optimize);78 79 enum llama_pooling_type pooling_type() const;80 81 float * get_logits();82 float * get_logits_ith(int32_t i);83 84 float * get_embeddings();85 float * get_embeddings_ith(int32_t i);86 float * get_embeddings_seq(llama_seq_id seq_id);87 88 float * get_embeddings_nextn();89 float * get_embeddings_nextn_ith(int32_t i);90 91 float * get_embeddings_layer_inp(uint32_t lid);92 93 llama_token * get_sampled_tokens() const;94 llama_token get_sampled_token_ith(int32_t idx);95 96 float * get_sampled_logits_ith(int32_t idx);97 size_t get_sampled_logits_count(int32_t idx);98 99 float * get_sampled_probs_ith(int32_t idx);100 size_t get_sampled_probs_count(int32_t idx);101 102 const llama_token * get_sampled_candidates_ith(int32_t idx);103 size_t get_sampled_candidates_count(int32_t idx);104 105 void attach_threadpool(106 ggml_threadpool_t threadpool,107 ggml_threadpool_t threadpool_batch);108 109 void detach_threadpool();110 111 void set_n_threads(int32_t n_threads, int32_t n_threads_batch);112 113 void set_abort_callback(bool (*abort_callback)(void * data), void * abort_callback_data);114 115 void set_embeddings (bool value);116 void set_embeddings_nextn(bool value, bool masked);117 void set_embeddings_layer_inp(uint32_t lid, bool enable);118 void set_nextn_layer_offset(int32_t offset);119 void set_causal_attn(bool value);120 void set_warmup(bool value);121 122 void set_adapters_lora(llama_adapter_lora ** adapters, size_t n_adapters, float * scales);123 124 bool adapters_lora_are_same(llama_adapter_lora ** adapters, size_t n_adapters, float * scales);125 126 bool set_adapter_cvec(127 const float * data,128 size_t len,129 int32_t n_embd,130 int32_t il_start,131 int32_t il_end);132 133 // process a single ubatch with a specific graph type134 // if memory_context is provided, it will be applied first to the context's memory135 // ret contains the status of the graph computation136 // returns nullptr only if ret != GGML_STATUS_SUCCESS137 llm_graph_result * process_ubatch(138 const llama_ubatch & ubatch,139 llm_graph_type gtype,140 llama_memory_context_i * mctx,141 ggml_status & ret);142 143 int encode(const llama_batch & batch_inp);144 int decode(const llama_batch & batch_inp);145 146 //147 // state save/load148 //149 150 size_t state_get_size();151 size_t state_get_data( uint8_t * dst, size_t size);152 size_t state_set_data(const uint8_t * src, size_t size);153 154 size_t state_seq_get_size(llama_seq_id seq_id, llama_state_seq_flags flags);155 156 size_t state_seq_get_data(llama_seq_id seq_id, uint8_t * dst, size_t size, llama_state_seq_flags flags);157 size_t state_seq_set_data(llama_seq_id seq_id, const uint8_t * src, size_t size, llama_state_seq_flags flags);158 159 bool state_load_file(160 const char * filepath,161 llama_token * tokens_out,162 size_t n_token_capacity,163 size_t * n_token_count_out);164 165 bool state_save_file(166 const char * filepath,167 const llama_token * tokens,168 size_t n_token_count);169 170 size_t state_seq_load_file(171 llama_seq_id seq_id,172 const char * filepath,173 llama_token * tokens_out,174 size_t n_token_capacity,175 size_t * n_token_count_out);176 177 size_t state_seq_save_file(178 llama_seq_id seq_id,179 const char * filepath,180 const llama_token * tokens,181 size_t n_token_count);182 183 //184 // perf185 //186 187 llama_perf_context_data perf_get_data() const;188 void perf_reset();189 190 llama_memory_breakdown memory_breakdown() const;191 192 //193 // training194 //195 196 void opt_init(struct llama_model * model, struct llama_opt_params lopt_params);197 198 // TODO: more flexible combinations of logical/physical batch size and context size199 void opt_epoch(200 ggml_opt_dataset_t dataset,201 ggml_opt_result_t result_train,202 ggml_opt_result_t result_eval,203 int64_t idata_split,204 ggml_opt_epoch_callback callback_train,205 ggml_opt_epoch_callback callback_eval);206 207 void opt_epoch_iter(208 ggml_opt_dataset_t dataset,209 ggml_opt_result_t result,210 const std::vector<llama_token> & tokens,211 const std::vector<llama_token> & labels_sparse,212 llama_batch & batch,213 ggml_opt_epoch_callback callback,214 bool train,215 int64_t idata_in_loop,216 int64_t ndata_in_loop,217 int64_t t_loop_start);218 219private:220 //221 // output222 //223 224 // Make sure enough space is available for outputs.225 // Returns max number of outputs for which space was reserved.226 uint32_t output_reserve(int32_t n_outputs);227 228 void output_reorder();229 230 // map the output row index `i` to batch index231 int64_t output_resolve_row(int32_t i) const;232 233 // async-copy enabled layer-input tensors (per cparams.output_layer_inp)234 // from backend into host-side embd_layer_inp buffers235 void extract_layer_inputs(const llm_graph_result * res, size_t token_offset, size_t n_tokens);236 237 //238 // graph239 //240 241public:242 uint32_t graph_max_nodes(uint32_t n_tokens) const;243 244 // can reuse the llm_graph_result instance of the context (for example to update a memory module)245 llm_graph_result * get_gf_res_reserve() const;246 247 // returns the result of ggml_backend_sched_graph_compute_async execution248 ggml_status graph_compute(ggml_cgraph * gf, bool batched);249 250 // reserve a graph with a dummy ubatch of the specified size251 ggml_cgraph * graph_reserve(252 uint32_t n_tokens, uint32_t n_seqs, uint32_t n_outputs, const llama_memory_context_i * mctx, bool split_only = false, size_t * sizes = nullptr);253 254 bool set_sampler(llama_seq_id seq_id, llama_sampler * sampler);255 256private:257 llm_graph_params graph_params(258 llm_graph_result * res,259 const llama_ubatch & ubatch,260 const llama_memory_context_i * mctx,261 llm_graph_type gtype) const;262 263 llm_graph_cb graph_get_cb() const;264 265 // disable auto fused ops (Flash Attention, Gated Delta Net) whose op lands on a device266 // that differs from the layer it belongs to (usually due to missing backend support)267 void resolve_fused_ops(const llama_memory_context_i * mctx, uint32_t n_seqs);268 269 // TODO: read/write lora adapters and cvec270 size_t state_write_data(llama_io_write_i & io);271 size_t state_read_data (llama_io_read_i & io);272 273 size_t state_seq_write_data(llama_io_write_i & io, llama_seq_id seq_id, llama_state_seq_flags flags);274 size_t state_seq_read_data (llama_io_read_i & io, llama_seq_id seq_id, llama_state_seq_flags flags);275 276 //277 // members278 //279 280 const llama_model & model;281 282 llama_cparams cparams;283 284 llama_adapter_cvec_ptr cvec;285 llama_adapter_loras_ptr loras;286 287 llama_cross cross; // TODO: tmp for handling cross-attention - need something better probably288 289 llama_memory_ptr memory;290 291 // decode output (2-dimensional array: [n_outputs][n_vocab])292 buffer_view<float> logits = {nullptr, 0};293 294 // embeddings output (2-dimensional array: [n_outputs][n_embd])295 // populated only when pooling_type == LLAMA_POOLING_TYPE_NONE296 buffer_view<float> embd = {nullptr, 0};297 298 // hidden state required by the nextn layers (2-dimensional array: [n_outputs][n_embd])299 // populated only when cparams.embeddings_nextn is enabled and the model graph300 // sets llm_graph_result::t_h_nextn301 buffer_view<float> embd_nextn = {nullptr, 0};302 303 // host buffers for output layer input embeddings, per layer304 // populated when cparams.output_layer_inp[il] is true305 std::vector<buffer_view<float>> embd_layer_inp;306 307 struct sampling_info {308 // !samplers.empty() to check if any samplers are active309 std::map<llama_seq_id, llama_sampler *> samplers;310 311 buffer_view<float> logits = {nullptr, 0};312 buffer_view<llama_token> sampled = {nullptr, 0};313 buffer_view<float> probs = {nullptr, 0};314 buffer_view<llama_token> candidates = {nullptr, 0};315 316 std::vector<uint32_t> logits_count;317 std::vector<uint32_t> probs_count;318 std::vector<uint32_t> candidates_count;319 320 // optimization321 std::vector<llama_token> token_ids_full_vocab;322 };323 324 sampling_info sampling;325 326 // sequence embeddings output (map of [n_embd] vectors)327 // populated only when pooling_type != LLAMA_POOLING_TYPE_NONE328 std::map<llama_seq_id, std::vector<float>> embd_seq;329 330 // reuse the batch_allocr to avoid unnecessary memory allocations331 std::unique_ptr<llama_batch_allocr> balloc;332 333 uint32_t n_outputs = 0; // number of actually-used outputs in the current ubatch or last logical batch334 335 std::vector<int32_t> output_ids; // map batch token positions to ids of the logits and embd buffers336 337 struct swap_info {338 uint32_t i0;339 uint32_t i1;340 };341 342 std::vector<swap_info> output_swaps;343 344 ggml_backend_sched_ptr sched;345 346 bool sched_need_reserve = true;347 348 ggml_backend_t backend_cpu = nullptr;349 std::vector<ggml_backend_ptr> backends;350 351 // training352 ggml_opt_context_t opt_ctx = nullptr;353 354 ggml_threadpool_t threadpool = nullptr;355 ggml_threadpool_t threadpool_batch = nullptr;356 357 ggml_abort_callback abort_callback = nullptr;358 void * abort_callback_data = nullptr;359 360 std::vector<std::pair<ggml_backend_t, ggml_backend_set_n_threads_t>> set_n_threads_fns;361 362 // pointers and buffer types used for the compute buffer of each backend363 std::vector<ggml_backend_t> backend_ptrs;364 std::vector<ggml_backend_buffer_type_t> backend_buft;365 std::vector<size_t> backend_buf_exp_size; // expected buffer sizes366 367 llm_graph_result_ptr gf_res_prev;368 llm_graph_result_ptr gf_res_reserve;369 370 // host buffer for the model output (logits and embeddings)371 ggml_backend_buffer_ptr buf_output;372 373 // keep copies of the per-sequence memory on the device374 std::map<llama_seq_id, llama_memory_buffers> mem_storage;375 376 bool has_evaluated_once = false;377 378 // env: LLAMA_GRAPH_REUSE_DISABLE379 bool graph_reuse_disable = false;380 381 // perf382 mutable int64_t t_start_us = 0;383 mutable int64_t t_load_us = 0;384 mutable int64_t t_p_eval_us = 0;385 mutable int64_t t_eval_us = 0;386 387 mutable int64_t t_compute_start_us = 0;388 mutable int64_t n_queued_tokens = 0;389 390 mutable int32_t n_p_eval = 0; // number of tokens in eval calls for the prompt (with batch size > 1)391 mutable int32_t n_eval = 0; // number of eval calls392 393 mutable int32_t n_reused = 0; // number of times the previous graph was reused394};395 