Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
llama-graph.h1340 linesDownload Raw Back to src
1#pragma once2 3#include "llama-arch.h"4#include "llama-batch.h"5#include "llama-hparams.h"6#include "llama-adapter.h"7 8#include <cstdint>9#include <vector>10#include <memory>11#include <set>12#include <functional>13#include <map>14 15struct ggml_cgraph;16struct ggml_context;17struct ggml_tensor;18 19struct llama_cparams;20struct llama_layer;21 22struct llama_memory_context_i;23 24class llama_kv_cache_context;25class llama_kv_cache_dsa_context;26class llama_kv_cache_msa_context;27class llama_kv_cache_dsv4_raw_context;28class llama_kv_cache_dsv4_context;29class llama_kv_cache_iswa_context;30class llama_memory_recurrent_context;31class llama_memory_hybrid_context;32class llama_memory_hybrid_iswa_context;33 34// certain models (typically multi-modal) can produce different types of graphs35enum llm_graph_type {36    LLM_GRAPH_TYPE_DEFAULT,37    LLM_GRAPH_TYPE_ENCODER,38    LLM_GRAPH_TYPE_DECODER,39    LLM_GRAPH_TYPE_DECODER_MTP,40};41 42enum llm_fused_op {43    LLM_FUSED_OP_FLASH_ATTN,44    LLM_FUSED_OP_GDN_AR,45    LLM_FUSED_OP_GDN_CH,46    LLM_FUSED_OP_LIGHTNING_INDEXER,47    LLM_FUSED_OP_DSV4_HC_PRE,48    LLM_FUSED_OP_DSV4_HC_COMB,49    LLM_FUSED_OP_DSV4_HC_POST,50};51 52enum llm_ffn_op_type : int {53    LLM_FFN_NONE = 0,           // sentinel: unset; archs must assign before use54    LLM_FFN_SILU,55    LLM_FFN_GELU,56    LLM_FFN_RELU,57    LLM_FFN_RELU_SQR,58    LLM_FFN_SWIGLU,59    LLM_FFN_GEGLU,60    LLM_FFN_REGLU,61    LLM_FFN_SWIGLU_OAI_MOE,62};63 64enum llm_ffn_gate_type {65    LLM_FFN_SEQ,66    LLM_FFN_PAR, // ffn_gate is parallel to ffn_up67};68 69enum llm_norm_type {70    LLM_NORM,71    LLM_NORM_RMS,72    LLM_NORM_GROUP,73};74 75// TODO: tmp - need something better to pass the data from the encoder to the decoder76struct llama_cross {77    // the output embeddings from the encoder as a ggml tensor78    // TODO: this needs more work to be correct, for now copy the embeddings data to host memory79    //       ref: https://github.com/ggml-org/llama.cpp/pull/11213#discussion_r196989252480    //ggml_tensor * t_embd = nullptr;81 82    int64_t n_embd = 0;83    int64_t n_enc  = 0;84 85    // embeddings data copied to host memory (tmp)86    std::vector<float> v_embd;87 88    // needed to construct the cross-attention mask in the decoder89    std::vector<std::set<llama_seq_id>> seq_ids_enc;90};91 92struct llm_graph_params;93 94//95// llm_graph_input96//97 98class llm_graph_input_i {99public:100    llm_graph_input_i() {101        const char * LLAMA_GRAPH_INPUT_DEBUG = getenv("LLAMA_GRAPH_INPUT_DEBUG");102        debug = LLAMA_GRAPH_INPUT_DEBUG ? atoi(LLAMA_GRAPH_INPUT_DEBUG) : 0;103    }104 105    virtual ~llm_graph_input_i() = default;106 107    virtual void set_input(const llama_ubatch * ubatch) = 0;108 109    // return true if the resulting input tensors using the provided graph parameters would be110    //   the same as the previous input tensors that we have currently stored in the object111    virtual bool can_reuse(const llm_graph_params & params) {112        // returning false here by default will prevent from reusing the graph if the check113        //   for the input type has not been implemented yet114        GGML_UNUSED(params);115        return false;116    }117protected:118    // env: LLAMA_GRAPH_INPUT_DEBUG119    int debug = 0;120};121 122using llm_graph_input_ptr = std::unique_ptr<llm_graph_input_i>;123 124class llm_graph_input_embd : public llm_graph_input_i {125public:126    llm_graph_input_embd(int64_t n_embd) : n_embd(n_embd) {}127    virtual ~llm_graph_input_embd() = default;128 129    void set_input(const llama_ubatch * ubatch) override;130 131    bool can_reuse(const llm_graph_params & params) override;132 133    ggml_tensor * tokens = nullptr; // I32 [n_batch]134    ggml_tensor * embd   = nullptr; // F32 [n_embd, n_batch]135 136    const int64_t n_embd = 0;137};138 139// similar to llm_graph_input_embd but with an additional hidden state input140class llm_graph_input_embd_h : public llm_graph_input_i {141public:142    llm_graph_input_embd_h(int64_t n_embd) : n_embd(n_embd) {}143    virtual ~llm_graph_input_embd_h() = default;144 145    void set_input(const llama_ubatch * ubatch) override;146 147    bool can_reuse(const llm_graph_params & params) override;148 149    ggml_tensor * tokens = nullptr; // I32 [n_batch]150    ggml_tensor * embd   = nullptr; // F32 [n_embd, n_batch]151    ggml_tensor * h      = nullptr; // F32 [n_embd, n_batch]152 153    const int64_t n_embd = 0;154};155 156class llm_graph_input_pos : public llm_graph_input_i {157public:158    llm_graph_input_pos(uint32_t n_pos_per_embd) : n_pos_per_embd(n_pos_per_embd) {}159    virtual ~llm_graph_input_pos() = default;160 161    void set_input(const llama_ubatch * ubatch) override;162 163    bool can_reuse(const llm_graph_params & params) override;164 165    ggml_tensor * pos = nullptr; // I32 [n_batch]166 167    const uint32_t n_pos_per_embd = 1;168};169 170// temperature tuning, used by llama4171class llm_graph_input_attn_temp : public llm_graph_input_i {172public:173    llm_graph_input_attn_temp(uint32_t n_attn_temp_floor_scale, float f_attn_temp_scale, float f_attn_temp_offset)174        : n_attn_temp_floor_scale(n_attn_temp_floor_scale), f_attn_temp_scale(f_attn_temp_scale), f_attn_temp_offset(f_attn_temp_offset) {}175    virtual ~llm_graph_input_attn_temp() = default;176 177    void set_input(const llama_ubatch * ubatch) override;178 179    ggml_tensor * attn_scale = nullptr; // F32 [n_batch]180 181    const uint32_t n_attn_temp_floor_scale;182    const float    f_attn_temp_scale;183    const float    f_attn_temp_offset;184};185 186class llm_graph_input_pos_bucket : public llm_graph_input_i {187public:188    llm_graph_input_pos_bucket(const llama_hparams & hparams) : hparams(hparams) {}189    virtual ~llm_graph_input_pos_bucket() = default;190 191    void set_input(const llama_ubatch * ubatch) override;192 193    ggml_tensor * pos_bucket = nullptr; // I32 [n_batch, n_batch]194 195    const llama_hparams hparams;196};197 198class llm_graph_input_pos_bucket_kv : public llm_graph_input_i {199public:200    llm_graph_input_pos_bucket_kv(201            const llama_hparams & hparams,202            const llama_kv_cache_context * mctx) : hparams(hparams), mctx(mctx) {}203    virtual ~llm_graph_input_pos_bucket_kv() = default;204 205    void set_input(const llama_ubatch * ubatch) override;206 207    ggml_tensor * pos_bucket = nullptr; // I32 [n_kv, n_batch]208 209    const llama_hparams hparams;210 211    const llama_kv_cache_context * mctx;212};213 214class llm_graph_input_out_ids : public llm_graph_input_i {215public:216    llm_graph_input_out_ids(217            const llama_hparams & hparams,218            const llama_cparams & cparams,219            uint32_t n_outputs) : hparams(hparams), cparams(cparams), n_outputs(n_outputs) {}220    virtual ~llm_graph_input_out_ids() = default;221 222    void set_input(const llama_ubatch * ubatch) override;223 224    bool can_reuse(const llm_graph_params & params) override;225 226    ggml_tensor * out_ids; // I32 [n_outputs]227 228    const llama_hparams hparams;229    const llama_cparams cparams;230 231    const uint32_t n_outputs;232};233 234class llm_graph_input_mean : public llm_graph_input_i {235public:236    llm_graph_input_mean(const llama_cparams & cparams) : cparams(cparams) {}237    virtual ~llm_graph_input_mean() = default;238 239    void set_input(const llama_ubatch * ubatch) override;240 241    ggml_tensor * mean; // F32 [n_batch, n_batch]242 243    const llama_cparams cparams;244};245 246class llm_graph_input_cls : public llm_graph_input_i {247public:248    llm_graph_input_cls(const llama_cparams & cparams, const llm_arch arch) : cparams(cparams), arch(arch) {}249    virtual ~llm_graph_input_cls() = default;250 251    void set_input(const llama_ubatch * ubatch) override;252 253    ggml_tensor * cls; // I32 [n_batch]254 255    const llama_cparams cparams;256    const llm_arch arch;257};258 259class llm_graph_input_rs : public llm_graph_input_i {260public:261    llm_graph_input_rs(const llama_memory_recurrent_context * mctx) : mctx(mctx) {}262    virtual ~llm_graph_input_rs() = default;263 264    void set_input(const llama_ubatch * ubatch) override;265 266    bool can_reuse(const llm_graph_params & params) override;267 268    ggml_tensor * s_copy;  // I32 [n_rs]269 270    // views of s_copy, computed once per graph271    // and shared across layers which use build_rs272    ggml_tensor * s_copy_main;   // I32 [n_seqs]273    ggml_tensor * s_copy_extra;  // I32 [n_rs - n_seqs]274 275    const llama_memory_recurrent_context * mctx;276 277    // used in view offsets, need to match for valid graph reuse278    uint32_t head;279    int32_t rs_z;280};281 282class llm_graph_input_cross_embd : public llm_graph_input_i {283public:284    llm_graph_input_cross_embd(285            const llama_cross * cross) : cross(cross) {}286    virtual ~llm_graph_input_cross_embd() = default;287 288    void set_input(const llama_ubatch * ubatch) override;289 290    ggml_tensor * cross_embd; // F32 [n_embd, n_outputs_enc]291 292    const llama_cross * cross;293};294 295class llm_graph_input_attn_no_cache : public llm_graph_input_i {296public:297    llm_graph_input_attn_no_cache(const llama_hparams & hparams, const llama_cparams & cparams) :298        hparams(hparams),299        cparams(cparams) {300    }301    ~llm_graph_input_attn_no_cache() = default;302 303    void set_input(const llama_ubatch * ubatch) override;304 305    ggml_tensor * get_kq_mask()     const { return self_kq_mask_cnv; }306    ggml_tensor * get_kq_mask_swa() const { return self_kq_mask_swa_cnv; }307 308    // n_tokens == n_batch309    ggml_tensor * self_kq_mask         = nullptr; // F32/F16 [n_tokens, n_batch/n_stream, 1, n_stream]310    ggml_tensor * self_kq_mask_cnv     = nullptr; //         [n_tokens, n_batch/n_stream, 1, n_stream]311    ggml_tensor * self_kq_mask_swa     = nullptr; // F32/F16 [n_tokens, n_batch/n_stream, 1, n_stream]312    ggml_tensor * self_kq_mask_swa_cnv = nullptr; //         [n_tokens, n_batch/n_stream, 1, n_stream]313 314    const llama_hparams hparams;315    const llama_cparams cparams;316};317 318class llm_graph_input_attn_kv : public llm_graph_input_i {319public:320    llm_graph_input_attn_kv(321            const llama_hparams & hparams,322            const llama_cparams & cparams,323            const llama_kv_cache_context * mctx) :324        hparams(hparams),325        cparams(cparams),326        mctx(mctx) {327    }328    ~llm_graph_input_attn_kv() = default;329 330    void set_input(const llama_ubatch * ubatch) override;331 332    bool can_reuse(const llm_graph_params & params) override;333 334    ggml_tensor * get_k_idxs() const { return self_k_idxs; }335    ggml_tensor * get_v_idxs() const { return self_v_idxs; }336 337    ggml_tensor * get_kq_mask() const { return self_kq_mask_cnv; }338 339    ggml_tensor * self_k_idxs = nullptr; // I64 [n_batch]340    ggml_tensor * self_v_idxs = nullptr; // I64 [n_batch] or [n_batch*n_embd_v_gqa]341 342    ggml_tensor * self_kq_mask     = nullptr; // F32/F16 [n_kv, n_batch/n_stream, 1, n_stream]343    ggml_tensor * self_kq_mask_cnv = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]344 345    // note: assumes v_rot^2 == I346    ggml_tensor * self_k_rot = nullptr;347    ggml_tensor * self_v_rot = nullptr;348 349    // note: these have to be copies because in order to be able to reuse a graph, its inputs350    //       need to carry these parameters with them. otherwise, they can point to freed351    //       llm_graph_params from a previous batch, causing stack-use-after-return352    const llama_hparams hparams;353    const llama_cparams cparams;354 355    const llama_kv_cache_context * mctx;356};357 358// V-less input for the KV cache359// ref: https://github.com/ggml-org/llama.cpp/pull/19067360class llm_graph_input_attn_k : public llm_graph_input_i {361public:362    llm_graph_input_attn_k(363            const llama_hparams & hparams,364            const llama_cparams & cparams,365            const llama_kv_cache_context * mctx) :366        hparams(hparams),367        cparams(cparams),368        mctx(mctx) {369    }370    ~llm_graph_input_attn_k() = default;371 372    void set_input(const llama_ubatch * ubatch) override;373 374    bool can_reuse(const llm_graph_params & params) override;375 376    ggml_tensor * get_k_idxs() const { return self_k_idxs; }377 378    ggml_tensor * get_kq_mask() const { return self_kq_mask_cnv; }379 380    ggml_tensor * self_k_idxs = nullptr; // I64 [n_batch]381 382    ggml_tensor * self_kq_mask     = nullptr; // F32/F16 [n_kv, n_batch/n_stream, 1, n_stream]383    ggml_tensor * self_kq_mask_cnv = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]384 385    const llama_hparams hparams;386    const llama_cparams cparams;387 388    const llama_kv_cache_context * mctx;389};390 391class llm_graph_input_attn_k_dsa : public llm_graph_input_i {392public:393    llm_graph_input_attn_k_dsa(394            const llama_hparams & hparams,395            const llama_cparams & cparams,396            const llama_kv_cache_dsa_context * mctx) :397        hparams(hparams),398        cparams(cparams),399        mctx(mctx) {400    }401    ~llm_graph_input_attn_k_dsa() = default;402 403    void set_input(const llama_ubatch * ubatch) override;404 405    bool can_reuse(const llm_graph_params & params) override;406 407    ggml_tensor * get_k_idxs_mla() const { return self_k_idxs_mla; }408    ggml_tensor * get_k_idxs_lid() const { return self_k_idxs_lid; }409 410    ggml_tensor * get_kq_mask_mla() const { return self_kq_mask_mla_cnv; }411    ggml_tensor * get_kq_mask_lid() const { return self_kq_mask_lid; }412 413    ggml_tensor * self_k_idxs_mla = nullptr; // I64 [n_batch]414    ggml_tensor * self_k_idxs_lid = nullptr; // I64 [n_batch]415 416    ggml_tensor * self_kq_mask_mla     = nullptr; // F32/F16 [n_kv, n_batch/n_stream, 1, n_stream]417    ggml_tensor * self_kq_mask_mla_cnv = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]418    ggml_tensor * self_kq_mask_lid     = nullptr; // F32     [n_kv, n_batch/n_stream, 1, n_stream]419    ggml_tensor * self_kq_mask_lid_cnv = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]420 421    ggml_tensor * self_k_rot_lid = nullptr;422 423    const llama_hparams hparams;424    const llama_cparams cparams;425 426    const llama_kv_cache_dsa_context * mctx;427};428 429// standard K/V attention input against the base cache, plus destination indices for the indexer key cache430class llm_graph_input_attn_kv_msa : public llm_graph_input_attn_kv {431public:432    llm_graph_input_attn_kv_msa(433            const llama_hparams & hparams,434            const llama_cparams & cparams,435            const llama_kv_cache_msa_context * mctx);436    ~llm_graph_input_attn_kv_msa() = default;437 438    void set_input(const llama_ubatch * ubatch) override;439 440    bool can_reuse(const llm_graph_params & params) override;441 442    ggml_tensor * get_k_idxs_idx() const { return self_k_idxs_idx; }443 444    ggml_tensor * self_k_idxs_idx = nullptr; // I64 [n_batch]445 446    const llama_kv_cache_msa_context * mctx_msa;447};448 449class llm_graph_input_attn_kv_iswa : public llm_graph_input_i {450public:451    llm_graph_input_attn_kv_iswa(452            const llama_hparams & hparams,453            const llama_cparams & cparams,454            const llama_kv_cache_iswa_context * mctx) :455        hparams(hparams),456        cparams(cparams),457        mctx(mctx) {458    }459    ~llm_graph_input_attn_kv_iswa() = default;460 461    void set_input(const llama_ubatch * ubatch) override;462 463    bool can_reuse(const llm_graph_params & params) override;464 465    ggml_tensor * get_k_idxs()     const { return self_k_idxs; }466    ggml_tensor * get_v_idxs()     const { return self_v_idxs; }467    ggml_tensor * get_k_idxs_swa() const { return self_k_idxs_swa; }468    ggml_tensor * get_v_idxs_swa() const { return self_v_idxs_swa; }469 470    ggml_tensor * get_kq_mask()     const { return self_kq_mask_cnv; }471    ggml_tensor * get_kq_mask_swa() const { return self_kq_mask_swa_cnv; }472 473    ggml_tensor * self_k_idxs     = nullptr; // I64 [n_batch]474    ggml_tensor * self_v_idxs     = nullptr; // I64 [n_batch] or [n_batch*n_embd_v_gqa]475    ggml_tensor * self_k_idxs_swa = nullptr; // I64 [n_batch]476    ggml_tensor * self_v_idxs_swa = nullptr; // I64 [n_batch] or [n_batch*n_embd_v_gqa]477 478    ggml_tensor * self_kq_mask         = nullptr; // F32/F16 [n_kv, n_batch/n_stream, 1, n_stream]479    ggml_tensor * self_kq_mask_cnv     = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]480    ggml_tensor * self_kq_mask_swa     = nullptr; // F32/F16 [n_kv, n_batch/n_stream, 1, n_stream]481    ggml_tensor * self_kq_mask_swa_cnv = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]482 483    ggml_tensor * self_k_rot = nullptr;484    ggml_tensor * self_v_rot = nullptr;485 486    ggml_tensor * self_k_rot_swa = nullptr;487    ggml_tensor * self_v_rot_swa = nullptr;488 489    const llama_hparams hparams;490    const llama_cparams cparams;491 492    const llama_kv_cache_iswa_context * mctx;493};494 495class llm_graph_input_attn_k_iswa : public llm_graph_input_i {496public:497    llm_graph_input_attn_k_iswa(498            const llama_hparams & hparams,499            const llama_cparams & cparams,500            const llama_kv_cache_iswa_context * mctx) :501        hparams(hparams),502        cparams(cparams),503        mctx(mctx) {504    }505    ~llm_graph_input_attn_k_iswa() = default;506 507    void set_input(const llama_ubatch * ubatch) override;508 509    bool can_reuse(const llm_graph_params & params) override;510 511    ggml_tensor * get_k_idxs()     const { return self_k_idxs; }512    ggml_tensor * get_k_idxs_swa() const { return self_k_idxs_swa; }513 514    ggml_tensor * get_kq_mask()     const { return self_kq_mask_cnv; }515    ggml_tensor * get_kq_mask_swa() const { return self_kq_mask_swa_cnv; }516 517    ggml_tensor * self_k_idxs     = nullptr; // I64 [n_batch]518    ggml_tensor * self_k_idxs_swa = nullptr; // I64 [n_batch]519 520    ggml_tensor * self_kq_mask         = nullptr; // F32/F16 [n_kv, n_batch/n_stream, 1, n_stream]521    ggml_tensor * self_kq_mask_cnv     = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]522    ggml_tensor * self_kq_mask_swa     = nullptr; // F32/F16 [n_kv, n_batch/n_stream, 1, n_stream]523    ggml_tensor * self_kq_mask_swa_cnv = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]524 525    ggml_tensor * self_k_rot = nullptr;526    ggml_tensor * self_k_rot_swa = nullptr;527 528    const llama_hparams hparams;529    const llama_cparams cparams;530 531    const llama_kv_cache_iswa_context * mctx;532};533 534// DSV4 raw graph inputs are SWA-only, but their mask may be stream-shaped535// so raw K can be concatenated with DSV4 compressed K in one attention op.536class llm_graph_input_dsv4_raw {537public:538    llm_graph_input_dsv4_raw(539            const llama_cparams & cparams,540            const llama_kv_cache_dsv4_raw_context * mctx) :541        cparams(cparams),542        mctx(mctx) {543    }544 545    void set_input(const llama_ubatch * ubatch);546 547    ggml_tensor * get_k_idxs() const { return self_k_idxs; }548    ggml_tensor * get_kq_mask() const { return self_kq_mask_cnv; }549 550    ggml_tensor * self_k_idxs = nullptr; // I64 [n_batch]551 552    ggml_tensor * self_kq_mask     = nullptr; // F32/F16 [n_kv, n_batch/n_stream, 1, n_stream]553    ggml_tensor * self_kq_mask_cnv = nullptr; //         [n_kv, n_batch/n_stream, 1, n_stream]554 555    ggml_tensor * self_k_rot = nullptr;556 557    const llama_cparams cparams;558 559    const llama_kv_cache_dsv4_raw_context * mctx;560};561 562class llm_graph_input_dsv4 : public llm_graph_input_i {563public:564    struct comp_input {565        ggml_tensor * state_pos        = nullptr; // I32 [n_state]566        ggml_tensor * state_persist_src_idxs = nullptr; // I32 [n_state_persist]567        ggml_tensor * state_persist_dst_idxs = nullptr; // I32 [n_state_persist]568        ggml_tensor * state_restore_src_idxs = nullptr; // I32 [n_state_restore]569        ggml_tensor * state_restore_dst_idxs = nullptr; // I32 [n_state_restore]570        ggml_tensor * state_snapshot_src_idxs = nullptr; // I32 [n_state_snapshot]571        ggml_tensor * state_snapshot_dst_idxs = nullptr; // I32 [n_state_snapshot]572        ggml_tensor * state_read_idxs  = nullptr; // I32 [ratio*n_state_write]573        ggml_tensor * state_write_idxs = nullptr; // I64 [n_state_write]574        ggml_tensor * state_write_pos  = nullptr; // I32 [n_state_write]575 576        ggml_tensor * kq_mask    = nullptr; // F32 [n_kv, n_batch/n_stream, 1, n_stream]577 578        ggml_tensor * k_rot      = nullptr;579    };580 581    llm_graph_input_dsv4(582            const llama_cparams & cparams,583            std::unique_ptr<llm_graph_input_dsv4_raw> inp_raw,584            const llama_kv_cache_dsv4_context * mctx) :585        inp_raw(std::move(inp_raw)),586        cparams(cparams),587        mctx(mctx) {588    }589    ~llm_graph_input_dsv4() = default;590 591    void set_input(const llama_ubatch * ubatch) override;592 593    bool can_reuse(const llm_graph_params & params) override;594 595    llm_graph_input_dsv4_raw * get_raw() const { return inp_raw.get(); }596    const comp_input & get_csa() const { return inp_csa; }597    const comp_input & get_hca() const { return inp_hca; }598    const comp_input & get_lid() const { return inp_lid; }599 600    std::unique_ptr<llm_graph_input_dsv4_raw> inp_raw;601 602    comp_input inp_csa;603    comp_input inp_hca;604    comp_input inp_lid;605 606    const llama_cparams cparams;607 608    const llama_kv_cache_dsv4_context * mctx;609};610 611class llm_graph_input_attn_cross : public llm_graph_input_i {612public:613    llm_graph_input_attn_cross(const llama_cross * cross) : cross(cross) {}614    ~llm_graph_input_attn_cross() = default;615 616    void set_input(const llama_ubatch * ubatch) override;617 618    ggml_tensor * get_kq_mask_cross() const { return cross_kq_mask_cnv; }619 620    ggml_tensor * cross_kq_mask     = nullptr; // F32/F16 [n_outputs_enc, n_batch, 1, 1]621    ggml_tensor * cross_kq_mask_cnv = nullptr; // F32/F16 [n_outputs_enc, n_batch, 1, 1]622 623    const llama_cross * cross = nullptr;624};625 626class llm_graph_input_mem_hybrid : public llm_graph_input_i {627public:628    llm_graph_input_mem_hybrid(629            const llama_cparams & cparams,630            std::unique_ptr<llm_graph_input_attn_kv> inp_attn,631            std::unique_ptr<llm_graph_input_rs>      inp_rs,632            const llama_memory_hybrid_context *      mctx) :633        inp_attn(std::move(inp_attn)),634        inp_rs(std::move(inp_rs)),635        cparams(cparams),636        mctx(mctx) { }637    virtual ~llm_graph_input_mem_hybrid() = default;638 639    void set_input(const llama_ubatch * ubatch) override;640 641    bool can_reuse(const llm_graph_params & params) override;642 643    std::unique_ptr<llm_graph_input_attn_kv> inp_attn;644    std::unique_ptr<llm_graph_input_rs>      inp_rs;645 646    llm_graph_input_attn_kv * get_attn() const { return inp_attn.get(); }647    llm_graph_input_rs      * get_recr() const { return inp_rs.get(); }648 649    const llama_cparams cparams;650 651    const llama_memory_hybrid_context * mctx;652};653 654class llm_graph_input_mem_hybrid_k : public llm_graph_input_i {655public:656    llm_graph_input_mem_hybrid_k(657            const llama_cparams & cparams,658            std::unique_ptr<llm_graph_input_attn_k> inp_attn,659            std::unique_ptr<llm_graph_input_rs>      inp_rs,660            const llama_memory_hybrid_context *      mctx) :661        inp_attn(std::move(inp_attn)),662        inp_rs(std::move(inp_rs)),663        cparams(cparams),664        mctx(mctx) { }665    virtual ~llm_graph_input_mem_hybrid_k() = default;666 667    void set_input(const llama_ubatch * ubatch) override;668 669    bool can_reuse(const llm_graph_params & params) override;670 671    std::unique_ptr<llm_graph_input_attn_k> inp_attn;672    std::unique_ptr<llm_graph_input_rs>      inp_rs;673 674    llm_graph_input_attn_k * get_attn() const { return inp_attn.get(); }675    llm_graph_input_rs      * get_recr() const { return inp_rs.get(); }676 677    const llama_cparams cparams;678 679    const llama_memory_hybrid_context * mctx;680};681 682class llm_graph_input_mem_hybrid_iswa : public llm_graph_input_i {683public:684    llm_graph_input_mem_hybrid_iswa(685            const llama_cparams & cparams,686            std::unique_ptr<llm_graph_input_attn_kv_iswa> inp_attn,687            std::unique_ptr<llm_graph_input_rs>          inp_rs,688            const llama_memory_hybrid_iswa_context *     mctx) :689        inp_attn(std::move(inp_attn)),690        inp_rs(std::move(inp_rs)),691        cparams(cparams),692        mctx(mctx) { }693    virtual ~llm_graph_input_mem_hybrid_iswa() = default;694 695    void set_input(const llama_ubatch * ubatch) override;696 697    bool can_reuse(const llm_graph_params & params) override;698 699    std::unique_ptr<llm_graph_input_attn_kv_iswa> inp_attn;700    std::unique_ptr<llm_graph_input_rs>          inp_rs;701 702    llm_graph_input_attn_kv_iswa * get_attn() const { return inp_attn.get(); }703    llm_graph_input_rs           * get_recr() const { return inp_rs.get(); }704 705    const llama_cparams cparams;706 707    const llama_memory_hybrid_iswa_context * mctx;708};709 710class llm_graph_input_sampling : public llm_graph_input_i {711public:712    llm_graph_input_sampling(std::map<llama_seq_id, llama_sampler *> samplers) :713        samplers(std::move(samplers)) { }714    virtual ~llm_graph_input_sampling() = default;715 716    void set_input(const llama_ubatch * ubatch) override;717    bool can_reuse(const llm_graph_params & params) override;718 719    std::map<llama_seq_id, llama_sampler *> samplers;720};721 722//723// llm_graph_result724//725 726// these objects deliver the result from the graph build process back to the llama_context727// note that the input tensors created for the graph are referenced here - the goal is to be able to populate their728//   specific data, by calling the set_inputs() method729// along with the input tensors, the object also provides commonly used outputs tensors, such as logits, embeddings, etc.730//   these are used by the llama_context to extact the relevant data, based on the compute parameters731 732// callback that allows us to apply custom logic to each tensor (e.g. ggml-alloc, offloading, etc.)733using llm_graph_cb = std::function<void(const llama_ubatch & ubatch, ggml_tensor * cur, const char * name, int il)>;734 735class llm_graph_result;736 737struct llm_graph_params {738    llm_arch arch = LLM_ARCH_UNKNOWN;739 740    llama_hparams hparams;741    llama_cparams cparams;742 743    llama_ubatch ubatch; // note: intentionally make a copy744 745    llm_graph_type gtype;746 747    ggml_backend_sched_t sched;748    ggml_backend_t backend_cpu;749 750    const llama_adapter_cvec     * cvec;751    const llama_adapter_loras    * loras;752    const llama_memory_context_i * mctx;753    const llama_cross            * cross;754 755    std::map<llama_seq_id, llama_sampler *> samplers;756 757    static bool samplers_equal(758          const std::map<llama_seq_id, llama_sampler *> & lhs,759          const std::map<llama_seq_id, llama_sampler *> & rhs) {760        if (lhs.size() != rhs.size()) {761            return false;762        }763        for (const auto & [seq_id, sampler] : lhs) {764            auto it = rhs.find(seq_id);765            if (it == rhs.end() || it->second != sampler) {766                return false;767            }768        }769        return true;770    }771 772    uint32_t n_outputs;773 774    llm_graph_cb cb;775 776    llm_graph_result * res;777 778    // return true if the "other" params would result in a graph with the same topology as with the current params779    //   having the same topology allows us to reuse the graph in some cases780    bool allow_reuse(const llm_graph_params & other) const {781        // first check the ubatch782        bool can_reuse_ubatch =783            ubatch.equal_seqs() == other.ubatch.equal_seqs() &&784            ubatch.n_tokens     == other.ubatch.n_tokens &&785            ubatch.n_seq_tokens == other.ubatch.n_seq_tokens &&786            ubatch.n_seqs       == other.ubatch.n_seqs &&787            ubatch.n_seqs_unq   == other.ubatch.n_seqs_unq &&788            (789                (!ubatch.token && !other.ubatch.token) ||790                (!ubatch.embd  && !other.ubatch.embd)  ||791                (ubatch.token && other.ubatch.token && ubatch.embd && other.ubatch.embd)792            );793 794        // when we split the batch using "equal_seqs" we have to verify that the participating sequences are the same795        //   the reason is because the set of attention streams would be different for different sequences796        if (can_reuse_ubatch && ubatch.equal_seqs()) {797            if (!ubatch.data) {798                // if the old ubatch does not own it's data, then we cannot guarantee that it is still alive, and799                //   therefore we cannot perform the sequence id check. normally should never happen800                can_reuse_ubatch = false;801            } else {802                for (uint32_t s = 0; s < ubatch.n_seqs_unq; ++s) {803                    can_reuse_ubatch &= ubatch.seq_id_unq[s] == other.ubatch.seq_id_unq[s];804                }805            }806        }807 808        if (!can_reuse_ubatch) {809            return false;810        }811 812        if (n_outputs != other.n_outputs) {813            return false;814        }815 816        if (!samplers_equal(samplers, other.samplers)) {817            return false;818        }819 820        if (samplers.size() > 0) {821            if (!ubatch.data || !other.ubatch.data) {822                return false;823            }824 825            // check that the outputs are the same for all samplers826            for (uint32_t i = 0; i < ubatch.n_tokens; ++i) {827                if (ubatch.output[i]    != other.ubatch.output[i] ||828                    ubatch.seq_id[i][0] != other.ubatch.seq_id[i][0]) {829                    return false;830                }831            }832        }833 834        // TODO: https://github.com/ggml-org/llama.cpp/pull/24340#discussion_r3448035248835        if (cparams.nextn_layer_offset != other.cparams.nextn_layer_offset) {836            return false;837        }838 839        return840            cparams.embeddings              == other.cparams.embeddings              &&841            cparams.embeddings_nextn        == other.cparams.embeddings_nextn        &&842            cparams.embeddings_nextn_masked == other.cparams.embeddings_nextn_masked &&843            cparams.causal_attn             == other.cparams.causal_attn             &&844            arch  == other.arch  &&845            gtype == other.gtype &&846            cvec  == other.cvec  &&847            loras == other.loras &&848            cross == other.cross;849    }850};851 852struct llm_graph_fused_node {853    llm_fused_op op;854    ggml_tensor * tensor;855    int il;856};857 858class llm_graph_result {859public:860    llm_graph_result(int64_t max_nodes);861 862    virtual ~llm_graph_result() = default;863 864    ggml_tensor * get_inp_tokens()  const { return t_inp_tokens; }865    ggml_tensor * get_logits()      const { return t_logits; }866    ggml_tensor * get_embd()        const { return t_embd; }867    ggml_tensor * get_embd_pooled() const { return t_embd_pooled; }868    ggml_tensor * get_h_nextn()     const { return t_h_nextn; }869 870    ggml_tensor * get_layer_inp(int il) const { return t_layer_inp[il]; }871 872    ggml_cgraph  * get_gf()  const { return gf; }873    ggml_context * get_ctx() const { return ctx_compute.get(); }874 875    int64_t get_max_nodes() const;876 877    void reset();878 879    void set_inputs(const llama_ubatch * ubatch);880    void set_outputs(const llm_graph_params & params);881 882    // try to update the existing graph result using the new graph parameters in order to reuse it883    // this can only be done if we determine that the resulting graph using the new graph parameters884    //   would be identical to the existing graph. in that case, we simply have to update the memory885    //   contexts of the input tensors of the graph and we can reuse it for another computation886    // return true if the graph was updated and can be reused887    bool can_reuse(const llm_graph_params & params);888 889    llm_graph_input_i * add_input(llm_graph_input_ptr input);890 891    void add_fused_node(llm_graph_fused_node result);892 893    const std::vector<llm_graph_fused_node> & get_fused_nodes() const { return fused_nodes; }894 895    void set_params(const llm_graph_params & params);896 897    // important graph nodes898    ggml_tensor * t_inp_tokens  = nullptr;899    ggml_tensor * t_inp_embd    = nullptr; // [n_embd_inp, n_tokens]900    ggml_tensor * t_logits      = nullptr;901    ggml_tensor * t_embd        = nullptr;902    ggml_tensor * t_embd_pooled = nullptr;903    ggml_tensor * t_h_nextn     = nullptr; // [n_embd, n_outputs] hidden state before final output norm904 905    std::vector<ggml_tensor *> t_layer_inp;906 907    std::vector<ggml_tensor *> t_sampled;908    std::vector<ggml_tensor *> t_sampled_probs;909    std::vector<ggml_tensor *> t_sampled_logits;910    std::vector<ggml_tensor *> t_candidates;911 912    std::vector<llm_graph_input_ptr> inputs;913    std::vector<llm_graph_fused_node> fused_nodes;914 915    ggml_context_ptr ctx_compute;916 917    // memory buffers used to evaluate the model918    std::vector<uint8_t> buf_compute_meta;919 920    ggml_cgraph * gf;921 922    int64_t max_nodes;923 924private:925    // keep a copy of the previous graph parameters926    // we will use this to determine whether the graph can be reused by comparing them with the new parameters927    // note: these are updated after constructing the new graph928    llm_graph_params params;929 930    // env: LLAMA_GRAPH_RESULT_DEBUG931    int debug = 0;932};933 934using llm_graph_result_ptr = std::unique_ptr<llm_graph_result>;935 936//937// llm_graph_context938//939 940// used in build_rs to properly order writes and avoid unnecessary copies941using llm_graph_get_rows_fn = std::function<ggml_tensor * (ggml_context *, ggml_tensor * states, ggml_tensor * ids)>;942 943struct llm_graph_qkv {944    ggml_tensor * q; // [n_embd_head, n_head,    n_tokens]945    ggml_tensor * k; // [n_embd_head, n_head_kv, n_tokens]946    ggml_tensor * v; // [n_embd_head, n_head_kv, n_tokens]947};948 949struct llm_graph_context {950    const llm_arch arch;951 952    const llama_hparams & hparams;953    const llama_cparams & cparams;954    const llama_ubatch  & ubatch;955 956    const int64_t n_embd;957    const int64_t n_layer;958    const int64_t n_layer_nextn;959    const int64_t n_rot;960    const int64_t n_ctx;       // user-specified context size (can be different from n_ctx_train)961    const int64_t n_head;962    const int64_t n_head_kv;963    const int64_t n_embd_head_k;964    const int64_t n_embd_k_gqa;965    const int64_t n_embd_head_v;966    const int64_t n_embd_v_gqa;967    const int64_t n_expert;968    const int64_t n_expert_used;969 970    const float freq_base;971    const float freq_scale;972    const float ext_factor;973    const float attn_factor;974    const float beta_fast;975    const float beta_slow;976    const float norm_eps;977    const float norm_rms_eps;978 979    const int64_t n_tokens;980    const int64_t n_outputs;981    const int32_t n_ctx_orig; // yarn982 983    const enum llama_pooling_type pooling_type;984    const enum llama_rope_type    rope_type;985 986    ggml_backend_sched_t sched;987 988    ggml_backend_t backend_cpu; // TODO: needed by build_attn_mha, figure out a way to remove?989 990    const llama_adapter_cvec     * cvec;991    const llama_adapter_loras    * loras;992    const llama_memory_context_i * mctx;993    const llama_cross            * cross;994 995    std::map<llama_seq_id, llama_sampler *> samplers;996 997    const llm_graph_cb & cb_func;998 999    llm_graph_result * res;1000 1001    ggml_context * ctx0 = nullptr;1002    ggml_cgraph  * gf   = nullptr;1003 1004    llm_graph_context(const llm_graph_params & params);1005    virtual ~llm_graph_context() = default;1006 1007    void cb(ggml_tensor * cur, const char * name, int il) const;1008 1009    //1010    // common1011    //1012 1013    ggml_tensor * build_cvec(1014             ggml_tensor * cur,1015                     int   il) const;1016 1017    // do mat_mul, while optionally apply lora and per-tensor scale1018    ggml_tensor * build_lora_mm(1019              ggml_tensor * w,1020              ggml_tensor * cur,1021              ggml_tensor * w_s = nullptr) const;1022 1023    // do mat_mul_id, while optionally apply lora and per-expert scale1024    ggml_tensor * build_lora_mm_id(1025              ggml_tensor * w,   // ggml_tensor * as1026              ggml_tensor * cur, // ggml_tensor * b1027              ggml_tensor * ids,1028              ggml_tensor * w_s = nullptr) const;1029 1030    ggml_tensor * build_norm(1031             ggml_tensor * cur,1032             ggml_tensor * mw,1033             ggml_tensor * mb,1034           llm_norm_type   type,1035                     int   il) const;1036 1037 1038    // compute Q, K, V projections with optional bias and reshape1039    // supports both fused wqkv and separate wq/wk/wv paths1040    llm_graph_qkv build_qkv(1041        const llama_layer & layer,1042              ggml_tensor * cur,1043                  int64_t   n_embd_head,1044                  int64_t   n_head,1045                  int64_t   n_head_kv,1046                      int   il) const;1047 1048    ggml_tensor * build_ffn(1049             ggml_tensor * cur,1050             ggml_tensor * up,1051             ggml_tensor * up_b,1052             ggml_tensor * up_s,1053             ggml_tensor * gate,1054             ggml_tensor * gate_b,1055             ggml_tensor * gate_s,1056             ggml_tensor * down,1057             ggml_tensor * down_b,1058             ggml_tensor * down_s,1059             ggml_tensor * act_scales,1060         llm_ffn_op_type   type_op,1061       llm_ffn_gate_type   type_gate,1062                     int   il) const;1063 1064    // build MoE FFN without bias tensors1065    ggml_tensor * build_moe_ffn(1066             ggml_tensor * cur,1067             ggml_tensor * gate_inp,1068             ggml_tensor * up_exps,1069             ggml_tensor * gate_exps,1070             ggml_tensor * down_exps,1071             ggml_tensor * exp_probs_b,1072                 int64_t   n_expert,1073                 int64_t   n_expert_used,1074         llm_ffn_op_type   type_op,1075                    bool   norm_w,1076                   float   w_scale,1077            llama_expert_gating_func_type gating_op,1078                     int   il,1079             ggml_tensor * probs_in = nullptr,1080             ggml_tensor * gate_up_exps = nullptr,1081             ggml_tensor * up_exps_s = nullptr,1082             ggml_tensor * gate_exps_s = nullptr,1083             ggml_tensor * down_exps_s = nullptr,1084             ggml_tensor * selected_experts_in = nullptr) const;1085 1086    ggml_tensor * build_moe_ffn(1087             ggml_tensor * cur,1088             ggml_tensor * gate_inp,1089             ggml_tensor * gate_inp_b,1090             ggml_tensor * up_exps,1091             ggml_tensor * up_exps_b,1092             ggml_tensor * gate_exps,1093             ggml_tensor * gate_exps_b,1094             ggml_tensor * down_exps,1095             ggml_tensor * down_exps_b,1096             ggml_tensor * exp_probs_b,1097                 int64_t   n_expert,1098                 int64_t   n_expert_used,1099         llm_ffn_op_type   type_op,1100                    bool   norm_w,1101                   float   w_scale,1102            llama_expert_gating_func_type gating_op,1103                     int   il,1104             ggml_tensor * probs_in = nullptr,1105             ggml_tensor * gate_up_exps = nullptr,1106             ggml_tensor * gate_up_exps_b = nullptr,1107             ggml_tensor * up_exps_s = nullptr,1108             ggml_tensor * gate_exps_s = nullptr,1109             ggml_tensor * down_exps_s = nullptr,1110             ggml_tensor * selected_experts_in = nullptr) const;1111 1112    //1113    // inputs1114    //1115 1116    ggml_tensor * build_inp_embd(ggml_tensor * tok_embd) const;1117    ggml_tensor * build_inp_pos() const;1118    ggml_tensor * build_inp_attn_scale() const;1119    ggml_tensor * build_inp_out_ids() const;1120    ggml_tensor * build_inp_mean() const;1121    ggml_tensor * build_inp_cls() const;1122 1123    ggml_tensor * build_inp_cross_embd() const;1124    ggml_tensor * build_inp_pos_bucket_enc() const;1125    ggml_tensor * build_inp_pos_bucket_dec() const;1126    ggml_tensor * build_pos_bias(ggml_tensor * pos_bucket, ggml_tensor * attn_rel_b) const;1127 1128    //1129    // attention1130    //1131 1132    ggml_tensor * build_attn_mha(1133            ggml_tensor * q,       // [n_embd_head_q, n_head_q, n_tokens]1134            ggml_tensor * k,       // [n_embd_head_k, n_head_k, n_tokens]1135            ggml_tensor * v,       // [n_embd_head_v, n_head_v, n_tokens] (v_trans = false)1136            ggml_tensor * kq_b,1137            ggml_tensor * kq_mask,1138            ggml_tensor * sinks,   // [n_head_q]1139            ggml_tensor * v_mla,   // [n_embd_head_v_mla, n_embd_head_v, n_head_v]1140                  float   kq_scale,1141                    int   il) const;1142 1143    llm_graph_input_attn_no_cache * build_attn_inp_no_cache() const;1144 1145    ggml_tensor * build_attn(1146            llm_graph_input_attn_no_cache * inp,1147            ggml_tensor * wo,1148            ggml_tensor * wo_b,1149            ggml_tensor * wo_s,1150            ggml_tensor * q_cur, // [n_embd_head_q, n_head_q, n_tokens]1151            ggml_tensor * k_cur, // [n_embd_head_k, n_head_k, n_tokens]1152            ggml_tensor * v_cur, // [n_embd_head_v, n_head_v, n_tokens]1153            ggml_tensor * kq_b,1154            ggml_tensor * sinks, // [n_head_q]1155            ggml_tensor * v_mla, // [n_embd_head_v_mla, n_embd_head_v, n_head_v]1156                  float   kq_scale,1157                    int   il) const;1158 1159    llm_graph_input_attn_kv * build_attn_inp_kv() const;1160 1161    ggml_tensor * build_attn(1162            llm_graph_input_attn_kv * inp,1163            ggml_tensor * wo,1164            ggml_tensor * wo_b,1165            ggml_tensor * wo_s,1166            ggml_tensor * q_cur, // [n_embd_head_q, n_head_q, n_tokens]1167            ggml_tensor * k_cur, // [n_embd_head_k, n_head_k, n_tokens]1168            ggml_tensor * v_cur, // [n_embd_head_v, n_head_v, n_tokens]1169            ggml_tensor * kq_b,1170            ggml_tensor * sinks, // [n_head_q]1171            ggml_tensor * v_mla, // [n_embd_head_v_mla, n_embd_head_v, n_head_v] // TODO: remove1172                  float   kq_scale,1173                    int   il) const;1174 1175    llm_graph_input_attn_k  * build_attn_inp_k() const;1176 1177    ggml_tensor * build_attn(1178            llm_graph_input_attn_k * inp,1179            ggml_tensor * wo,1180            ggml_tensor * wo_b,1181            ggml_tensor * wo_s,1182            ggml_tensor * q_cur, // [n_embd_head_q, n_head_q, n_tokens]1183            ggml_tensor * k_cur, // [n_embd_head_k, n_head_k, n_tokens]1184            ggml_tensor * v_cur, // [n_embd_head_v, n_head_v, n_tokens]1185            ggml_tensor * kq_b,1186            ggml_tensor * sinks, // [n_head_q]1187            ggml_tensor * v_mla, // [n_embd_head_v_mla, n_embd_head_v, n_head_v]1188                  float   kq_scale,1189                    int   il) const;1190 1191    llm_graph_input_attn_k_dsa * build_attn_inp_k_dsa() const;1192 1193    llm_graph_input_attn_kv_msa * build_attn_inp_kv_msa(bool msa_enabled) const;1194 1195    ggml_tensor * build_attn(1196            llm_graph_input_attn_k_dsa * inp,1197            ggml_tensor * wo,1198            ggml_tensor * wo_b,1199            ggml_tensor * wo_s,1200            ggml_tensor * q_cur, // [n_embd_head_q, n_head_q, n_tokens]

Showing the first 1,200 of 1340 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai