Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
dflash.cpp676 linesDownload Raw Back to models
1#include "models.h"2 3#include "llama-impl.h"4#include "llama-kv-cache.h"5#include "llama-kv-cache-iswa.h"6 7void llama_model_dflash::load_arch_hparams(llama_model_loader & ml) {8 9    ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);10 11    if (!ml.get_arr(LLM_KV_TARGET_LAYERS, target_layer_ids, false)) {12        throw std::runtime_error("DFlash model requires 'target_layers' in GGUF metadata");13    }14 15    hparams.n_embd_inp_enc_impl = (uint32_t) target_layer_ids.size() * hparams.n_embd;16 17    LLAMA_LOG_INFO("%s: DFlash extract_layers = [", __func__);18    for (size_t i = 0; i < target_layer_ids.size(); ++i) {19        LLAMA_LOG_INFO("%d%s", target_layer_ids[i], i + 1 < target_layer_ids.size() ? ", " : "");20    }21    LLAMA_LOG_INFO("]\n");22 23    // DeepSeek-V4 DSpark backbone: stages are full DSV4 blocks, uniform sliding window (the draft KV ring)24    ml.get_key(LLM_KV_HYPER_CONNECTION_COUNT, hparams.dsv4_hc_mult, false);25    if (hparams.dsv4_hc_mult > 0) {26        ml.get_key(LLM_KV_ATTENTION_Q_LORA_RANK,                hparams.n_lora_q);27        ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW,             hparams.n_swa);28        ml.get_key(LLM_KV_EXPERT_FEED_FORWARD_LENGTH,           hparams.n_ff_exp);29        ml.get_key(LLM_KV_EXPERT_SHARED_COUNT,                  hparams.n_expert_shared);30        ml.get_key(LLM_KV_EXPERT_WEIGHTS_SCALE,                 hparams.expert_weights_scale);31        ml.get_key(LLM_KV_EXPERT_WEIGHTS_NORM,                  hparams.expert_weights_norm);32        ml.get_key(LLM_KV_EXPERT_GATING_FUNC,                   hparams.expert_gating_func);33        ml.get_key_or_arr(LLM_KV_SWIGLU_CLAMP_EXP,              hparams.swiglu_clamp_exp, hparams.n_layer_all);34        if (!ml.get_key_or_arr(LLM_KV_SWIGLU_CLAMP_SHEXP,       hparams.swiglu_clamp_shexp, hparams.n_layer_all, 0)) {35            hparams.swiglu_clamp_shexp = hparams.swiglu_clamp_exp;36        }37        ml.get_key(LLM_KV_ATTENTION_OUTPUT_GROUP_COUNT,         hparams.dsv4_o_group_count);38        ml.get_key(LLM_KV_ATTENTION_OUTPUT_LORA_RANK,           hparams.dsv4_o_lora_rank);39        ml.get_key(LLM_KV_HYPER_CONNECTION_SINKHORN_ITERATIONS, hparams.dsv4_hc_sinkhorn_iters);40        ml.get_key(LLM_KV_HYPER_CONNECTION_EPSILON,             hparams.dsv4_hc_eps);41        ml.get_arr(LLM_KV_ATTENTION_COMPRESS_RATIOS,            hparams.dsv4_compress_ratios, false);42 43        if (hparams.expert_gating_func != LLAMA_EXPERT_GATING_FUNC_TYPE_SQRT_SOFTPLUS) {44            throw std::runtime_error("DSpark DSV4 draft expects sqrtsoftplus MoE scoring");45        }46        for (uint32_t il = 0; il < hparams.n_layer_all; ++il) {47            if (hparams.dsv4_compress_ratios[il] != 0) {48                throw std::runtime_error("DSpark DSV4 draft expects uncompressed attention on all stages");49            }50        }51 52        GGML_ASSERT(hparams.n_swa > 0);53        hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;54        hparams.set_swa_pattern(0);55        for (uint32_t il = 0; il < hparams.n_layer_all; ++il) {56            hparams.is_swa_impl[il] = true;57        }58        hparams.rope_freq_base_train_swa  = hparams.rope_freq_base_train;59        hparams.rope_freq_scale_train_swa = hparams.rope_freq_scale_train;60 61        type = LLM_TYPE_UNKNOWN;62        return;63    }64 65    // optional interleaved sliding-window attention with per-layer pattern array.66    // DFlash has a single rope, so the SWA rope == main rope.67    if (ml.get_key(LLM_KV_ATTENTION_SLIDING_WINDOW, hparams.n_swa, false) && hparams.n_swa > 0) {68        hparams.swa_type = LLAMA_SWA_TYPE_STANDARD;69        ml.get_key_or_arr(LLM_KV_ATTENTION_SLIDING_WINDOW_PATTERN, hparams.is_swa_impl, hparams.n_layer());70        hparams.rope_freq_base_train_swa  = hparams.rope_freq_base_train;71        hparams.rope_freq_scale_train_swa = hparams.rope_freq_scale_train;72    }73 74    type = LLM_TYPE_UNKNOWN;75}76 77void llama_model_dflash::load_arch_tensors(llama_model_loader &) {78    LLAMA_LOAD_LOCALS;79 80    const int64_t n_embd_inp = hparams.n_embd_inp_enc();81 82    // DSpark = DFlash + a semi-autoregressive Markov head and Confidence head83    //84    // TODO: only Qwen3-style backbones are supported for now; other backbones (e.g. Gemma4)85    //       need their own conversion path and graph tweaks86    const struct ggml_tensor * markov_meta = ml->get_tensor_meta("markov_w1.weight");87    if (markov_meta) {88        const int64_t dspark_markov_rank = markov_meta->ne[0];89 90        dspark_markov_w1 = create_tensor(tn(LLM_TENSOR_DSPARK_MARKOV_W1, "weight"), { dspark_markov_rank, n_vocab }, 0);91        dspark_markov_w2 = create_tensor(tn(LLM_TENSOR_DSPARK_MARKOV_W2, "weight"), { dspark_markov_rank, n_vocab }, 0);92 93        dspark_conf_proj   = create_tensor(tn(LLM_TENSOR_DSPARK_CONF_PROJ, "weight"), { n_embd + dspark_markov_rank, 1 }, 0);94        dspark_conf_proj_b = create_tensor(tn(LLM_TENSOR_DSPARK_CONF_PROJ, "bias"),   { 1 },             TENSOR_NOT_REQUIRED);95 96        LLAMA_LOG_INFO("%s: DFlash with DSpark markov head (rank = %lld)\n", __func__, (long long) dspark_markov_rank);97    }98 99    fc              = create_tensor(tn(LLM_TENSOR_FC,              "weight"), { n_embd_inp, n_embd }, 0);100    output_norm_enc = create_tensor(tn(LLM_TENSOR_ENC_OUTPUT_NORM, "weight"), { n_embd }, 0); // encoder hidden_norm (after fc)101    output_norm     = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM,    "weight"), { n_embd }, 0); // decoder final norm102 103    if (hparams.dsv4_hc_mult > 0) {104        const int64_t q_lora_rank     = hparams.n_lora_q;105        const int64_t n_ff_exp        = hparams.n_ff_exp;106        const int64_t n_expert_shared = hparams.n_expert_shared;107        const int64_t n_embd_head     = hparams.n_embd_head_k();108        const int64_t o_groups        = hparams.dsv4_o_group_count;109        const int64_t o_lora_rank     = hparams.dsv4_o_lora_rank;110        const int64_t hc_mult         = hparams.dsv4_hc_mult;111        const int64_t hc_dim          = hc_mult * n_embd;112        const int64_t hc_mix_dim      = (2 + hc_mult) * hc_mult;113 114        hc_head_fn    = create_tensor(tn(LLM_TENSOR_HC_HEAD_FN,    "weight"), {hc_dim, hc_mult}, 0);115        hc_head_base  = create_tensor(tn(LLM_TENSOR_HC_HEAD_BASE,  "weight"), {hc_mult}, 0);116        hc_head_scale = create_tensor(tn(LLM_TENSOR_HC_HEAD_SCALE, "weight"), {1}, 0);117 118        for (int i = 0; i < n_layer; ++i) {119            auto & layer = layers[i];120 121            layer.attn_norm     = create_tensor(tn(LLM_TENSOR_ATTN_NORM,     "weight", i), {n_embd}, 0);122            layer.attn_sinks    = create_tensor(tn(LLM_TENSOR_ATTN_SINKS,    "weight", i), {n_head}, 0);123            layer.wq_a          = create_tensor(tn(LLM_TENSOR_ATTN_Q_A,      "weight", i), {n_embd, q_lora_rank}, 0);124            layer.attn_q_a_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_A_NORM, "weight", i), {q_lora_rank}, 0);125            layer.wq_b          = create_tensor(tn(LLM_TENSOR_ATTN_Q_B,      "weight", i), {q_lora_rank, n_head * n_embd_head}, 0);126            layer.wkv           = create_tensor(tn(LLM_TENSOR_ATTN_KV,       "weight", i), {n_embd, n_embd_head}, 0);127            layer.attn_kv_norm  = create_tensor(tn(LLM_TENSOR_ATTN_KV_NORM,  "weight", i), {n_embd_head}, 0);128            layer.wo_a          = create_tensor(tn(LLM_TENSOR_ATTN_OUT_A,    "weight", i), {n_head * n_embd_head / o_groups, o_lora_rank, o_groups}, TENSOR_ALLOW_RESHAPE);129            layer.wo_b          = create_tensor(tn(LLM_TENSOR_ATTN_OUT_B,    "weight", i), {o_groups * o_lora_rank, n_embd}, 0);130 131            layer.hc_attn_fn    = create_tensor(tn(LLM_TENSOR_HC_ATTN_FN,    "weight", i), {hc_dim, hc_mix_dim}, 0);132            layer.hc_attn_base  = create_tensor(tn(LLM_TENSOR_HC_ATTN_BASE,  "weight", i), {hc_mix_dim}, 0);133            layer.hc_attn_scale = create_tensor(tn(LLM_TENSOR_HC_ATTN_SCALE, "weight", i), {3}, 0);134            layer.hc_ffn_fn     = create_tensor(tn(LLM_TENSOR_HC_FFN_FN,     "weight", i), {hc_dim, hc_mix_dim}, 0);135            layer.hc_ffn_base   = create_tensor(tn(LLM_TENSOR_HC_FFN_BASE,   "weight", i), {hc_mix_dim}, 0);136            layer.hc_ffn_scale  = create_tensor(tn(LLM_TENSOR_HC_FFN_SCALE,  "weight", i), {3}, 0);137 138            layer.ffn_gate_inp    = create_tensor(tn(LLM_TENSOR_FFN_GATE_INP,    "weight", i), {n_embd, n_expert}, 0);139            layer.ffn_exp_probs_b = create_tensor(tn(LLM_TENSOR_FFN_EXP_PROBS_B, "bias",   i), {n_expert}, 0);140            layer.ffn_norm        = create_tensor(tn(LLM_TENSOR_FFN_NORM,        "weight", i), {n_embd}, 0);141 142            layer.ffn_gate_exps = create_tensor(tn(LLM_TENSOR_FFN_GATE_EXPS, "weight", i), {n_embd,   n_ff_exp, n_expert}, 0);143            layer.ffn_down_exps = create_tensor(tn(LLM_TENSOR_FFN_DOWN_EXPS, "weight", i), {n_ff_exp, n_embd,   n_expert}, 0);144            layer.ffn_up_exps   = create_tensor(tn(LLM_TENSOR_FFN_UP_EXPS,   "weight", i), {n_embd,   n_ff_exp, n_expert}, 0);145 146            layer.ffn_gate_shexp = create_tensor(tn(LLM_TENSOR_FFN_GATE_SHEXP, "weight", i), {n_embd,                     n_ff_exp * n_expert_shared}, 0);147            layer.ffn_down_shexp = create_tensor(tn(LLM_TENSOR_FFN_DOWN_SHEXP, "weight", i), {n_ff_exp * n_expert_shared, n_embd                    }, 0);148            layer.ffn_up_shexp   = create_tensor(tn(LLM_TENSOR_FFN_UP_SHEXP,   "weight", i), {n_embd,                     n_ff_exp * n_expert_shared}, 0);149        }150        return;151    }152 153    for (int i = 0; i < n_layer; ++i) {154        auto & layer = layers[i];155 156        layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), { n_embd }, 0);157 158        layer.wq = create_tensor(tn(LLM_TENSOR_ATTN_Q,   "weight", i), { n_embd, n_embd_head_k * n_head }, 0);159        layer.wk = create_tensor(tn(LLM_TENSOR_ATTN_K,   "weight", i), { n_embd, n_embd_k_gqa }, 0);160        layer.wv = create_tensor(tn(LLM_TENSOR_ATTN_V,   "weight", i), { n_embd, n_embd_v_gqa }, 0);161        layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), { n_embd_head_k * n_head, n_embd }, 0);162 163        layer.attn_q_norm = create_tensor(tn(LLM_TENSOR_ATTN_Q_NORM, "weight", i), { n_embd_head_k }, 0);164        layer.attn_k_norm = create_tensor(tn(LLM_TENSOR_ATTN_K_NORM, "weight", i), { n_embd_head_k }, 0);165 166        layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), { n_embd }, 0);167        layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), { n_embd, n_ff }, 0);168        layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), { n_ff, n_embd }, 0);169        layer.ffn_up   = create_tensor(tn(LLM_TENSOR_FFN_UP,   "weight", i), { n_embd, n_ff }, 0);170    }171}172 173std::unique_ptr<llm_graph_context> llama_model_dflash::build_arch_graph(const llm_graph_params & params) const {174    switch (params.gtype) {175        case LLM_GRAPH_TYPE_ENCODER:176            return std::make_unique<graph<true>>(*this, params);177        case LLM_GRAPH_TYPE_DEFAULT:178        case LLM_GRAPH_TYPE_DECODER:179            if (hparams.dsv4_hc_mult > 0) {180                return std::make_unique<graph_dsv4>(*this, params);181            }182            return std::make_unique<graph<false>>(*this, params);183        default:184            GGML_ABORT("invalid graph type");185    };186}187 188template <>189ggml_tensor * llama_model_dflash::graph<true>::build_inp_embd_enc() const {190    auto inp_target = std::make_unique<llm_graph_input_embd>(hparams.n_embd_inp_enc());191 192    inp_target->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, hparams.n_embd_inp_enc(), n_tokens);193    ggml_set_input(inp_target->embd);194 195    ggml_tensor * cur = inp_target->embd;196    cb(cur, "inp_embd", -1);197 198    res->add_input(std::move(inp_target));199 200    return cur;201}202 203// DFlash Encoder: processes target model features through feature fusion layer204template <>205llama_model_dflash::graph<true>::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) {206    ggml_tensor * cur = build_inp_embd_enc();207 208    cur = build_lora_mm(model.fc, cur);209    cb(cur, "fc_out", -1);210 211    cur = build_norm(cur, model.output_norm_enc, NULL, LLM_NORM_RMS, -1);212    cb(cur, "enc_norm_out", -1);213 214    ggml_set_output(cur);215    res->t_h_nextn = cur;216 217    ggml_build_forward_expand(gf, cur);218}219 220// DSpark (DFlash + Markov & Confidence head): Markov bias on the draft logits, chained per block position221static void build_dspark_markov_head(llm_graph_context & g, const llama_model & model, ggml_tensor * tokens) {222    ggml_context * ctx0 = g.ctx0;223    auto         & res  = g.res;224 225    ggml_tensor * w1 = model.dspark_markov_w1;226    ggml_tensor * w2 = model.dspark_markov_w2;227    GGML_ASSERT(w1 && w2 && model.dspark_conf_proj && "DSpark markov/confidence weights not loaded");228 229    ggml_tensor * base = res->t_logits; // [n_vocab, n_tokens]230    const int64_t n_vocab = base->ne[0];231    const int64_t n_tok   = base->ne[1];232 233    const auto it = model.gguf_kv.find("dflash.block_size");234    GGML_ASSERT(it != model.gguf_kv.end() && "DSpark draft requires 'dflash.block_size' in GGUF metadata");235    const int64_t block_size = std::stoi(it->second);236    GGML_ASSERT(block_size > 0);237 238    const int64_t n_blocks = g.ubatch.n_seqs_unq;239    GGML_ASSERT(n_blocks > 0 && n_tok % n_blocks == 0 && "DSpark markov head requires equal-size blocks");240    // runtime tokens per block in this ubatch (anchor + drafted positions), bounded by training block_size241    const int64_t block_drafts = n_tok / n_blocks;242    if (block_drafts > block_size) {243        return;244    }245 246    // anchor (committed last) token of every block: token 0 of each block, i.e. a strided view247    const size_t token_stride = (size_t) block_drafts * tokens->nb[0];248    const size_t base_stride = (size_t) block_drafts * base->nb[1];249 250    ggml_tensor * prev = ggml_view_2d(ctx0, tokens, 1, n_blocks, token_stride, 0);251    prev = ggml_cont_1d(ctx0, prev, n_blocks);252 253    // confidence head input: predicts per-position acceptance254    ggml_tensor * conf_inp = res->t_embd; // [n_embd, n_tok]255 256    ggml_tensor * cat      = nullptr;257    ggml_tensor * cat_conf = nullptr;258 259    // TODO: the in-graph chain is greedy (argmax); sampling params affect only the final260    //       token pick, not the Markov conditioning path261    for (int64_t i = 0; i < block_drafts; ++i) {262        ggml_tensor * w1_prev = ggml_get_rows(ctx0, w1, prev);   // [R, n_blocks]263        ggml_tensor * bias    = ggml_mul_mat(ctx0, w2, w1_prev); // [n_vocab, n_blocks]264 265        // position i of every block: strided view [n_vocab, n_blocks]266        ggml_tensor * base_i = ggml_view_2d(ctx0, base, n_vocab, n_blocks, base_stride, i*base->nb[1]);267        ggml_tensor * col    = ggml_add(ctx0, base_i, bias);268 269        cat = cat ? ggml_concat(ctx0, cat, col, 1) : col;270 271        // conf(i) = sigmoid(conf_proj . [conf_inp(i); markov_w1[prev(i)]] + b)  -- [1, n_blocks]272        ggml_tensor * conf_inp_i = ggml_view_2d(ctx0, conf_inp, conf_inp->ne[0], n_blocks,273                                                (size_t) block_drafts * conf_inp->nb[1], i*conf_inp->nb[1]);274        ggml_tensor * feat = ggml_concat(ctx0, ggml_cont(ctx0, conf_inp_i), w1_prev, 0);275        ggml_tensor * conf = ggml_mul_mat(ctx0, model.dspark_conf_proj, feat);276        if (model.dspark_conf_proj_b) {277            conf = ggml_add(ctx0, conf, model.dspark_conf_proj_b);278        }279        conf = ggml_sigmoid(ctx0, conf);280 281        cat_conf = cat_conf ? ggml_concat(ctx0, cat_conf, conf, 1) : conf;282 283        if (i + 1 < block_drafts) {284            prev = ggml_argmax(ctx0, col);285        }286    }287 288    // cat is position-major; restore ubatch block-major order289    ggml_tensor * out = ggml_reshape_3d(ctx0, cat, n_vocab, n_blocks, block_drafts);290    out = ggml_cont(ctx0, ggml_permute(ctx0, out, 0, 2, 1, 3)); // [n_vocab, block_drafts, n_blocks]291    out = ggml_reshape_2d(ctx0, out, n_vocab, n_tok);292 293    {294        ggml_tensor * conf = ggml_reshape_3d(ctx0, cat_conf, 1, n_blocks, block_drafts);295        conf = ggml_cont(ctx0, ggml_permute(ctx0, conf, 0, 2, 1, 3));296        conf = ggml_reshape_2d(ctx0, conf, 1, n_tok);297 298        // note: broadcast the [1, n_tok] confidences to n_embd-wide rows to be able to reuse `llama_get_embeddings_nextn`299        conf = ggml_repeat(ctx0, conf, res->t_embd);300        res->t_h_nextn = conf;301        ggml_build_forward_expand(g.gf, conf);302    }303 304    res->t_logits = out;305    ggml_build_forward_expand(g.gf, out);306}307 308// DFlash decoder, dual-mode by batch type:309//   * embd batch  -> fused target features: project + inject K/V into the cache.310//   * token batch -> noise-block diffusion: attend over [committed, MASK...] to generate draft tokens311template <>312llama_model_dflash::graph<false>::graph(const llama_model & model, const llm_graph_params & params) : llm_graph_context(params) {313    const int64_t n_embd_head = hparams.n_embd_head_v();314 315    GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());316 317    ggml_tensor * inp_pos  = build_inp_pos();318 319    // optional iSWA: pick the matching attention input320    const bool use_iswa = hparams.swa_type != LLAMA_SWA_TYPE_NONE;321 322    llm_graph_input_attn_kv      * inp_attn      = nullptr;323    llm_graph_input_attn_kv_iswa * inp_attn_iswa = nullptr;324    if (use_iswa) {325        inp_attn_iswa = build_attn_inp_kv_iswa();326    } else {327        inp_attn = build_attn_inp_kv();328    }329 330    const float kq_scale = 1.0f/sqrtf(float(n_embd_head));331 332    // KV cache injection333    if (ubatch.embd) {334        auto inp = std::make_unique<llm_graph_input_embd>(n_embd);335 336        inp->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_tokens);337        ggml_set_input(inp->embd);338 339        ggml_tensor * inp_g = inp->embd;340        cb(inp_g, "inp_g_embeddings", -1);341 342        res->add_input(std::move(inp));343 344        for (int il = 0; il < n_layer; ++il) {345            const auto & layer = model.layers[il];346 347            ggml_tensor * Kcur = build_lora_mm(layer.wk, inp_g);348            ggml_tensor * Vcur = build_lora_mm(layer.wv, inp_g);349 350            Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);351            Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);352 353            Kcur = build_norm(Kcur, layer.attn_k_norm, NULL, LLM_NORM_RMS, il);354            Kcur = ggml_rope_ext(355                    ctx0, Kcur, inp_pos, nullptr,356                    n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,357                    ext_factor, attn_factor, beta_fast, beta_slow358                    );359            cb(Kcur, "Kcur_injected", il);360            cb(Vcur, "Vcur_injected", il);361 362            if (use_iswa) {363                // route each layer's K/V to its sub-cache: SWA layers -> sliding cache, full -> dense364                const bool    is_swa = hparams.is_swa(il);365                const auto  * kv     = is_swa ? inp_attn_iswa->mctx->get_swa() : inp_attn_iswa->mctx->get_base();366                ggml_tensor * k_idxs = is_swa ? inp_attn_iswa->get_k_idxs_swa() : inp_attn_iswa->get_k_idxs();367                ggml_tensor * v_idxs = is_swa ? inp_attn_iswa->get_v_idxs_swa() : inp_attn_iswa->get_v_idxs();368                // rotate K/V into the cache's rotated space369                ggml_tensor * k_rot  = is_swa ? inp_attn_iswa->self_k_rot_swa : inp_attn_iswa->self_k_rot;370                ggml_tensor * v_rot  = is_swa ? inp_attn_iswa->self_v_rot_swa : inp_attn_iswa->self_v_rot;371                if (k_rot) {372                    Kcur = llama_mul_mat_hadamard(ctx0, Kcur, k_rot);373                }374                if (v_rot) {375                    Vcur = llama_mul_mat_hadamard(ctx0, Vcur, v_rot);376                }377                ggml_build_forward_expand(gf, kv->cpy_k(ctx0, Kcur, k_idxs, il));378                ggml_build_forward_expand(gf, kv->cpy_v(ctx0, Vcur, v_idxs, il));379            } else {380                // rotate K/V into the cache's rotated space381                if (inp_attn->self_k_rot) {382                    Kcur = llama_mul_mat_hadamard(ctx0, Kcur, inp_attn->self_k_rot);383                }384                if (inp_attn->self_v_rot) {385                    Vcur = llama_mul_mat_hadamard(ctx0, Vcur, inp_attn->self_v_rot);386                }387                ggml_build_forward_expand(gf, inp_attn->mctx->cpy_k(ctx0, Kcur, inp_attn->get_k_idxs(), il));388                ggml_build_forward_expand(gf, inp_attn->mctx->cpy_v(ctx0, Vcur, inp_attn->get_v_idxs(), il));389            }390        }391 392        res->t_embd = inp_g;393 394        ggml_build_forward_expand(gf, inp_g);395        return;396    }397 398    // tok_embd from the target model (shared via ctx_other)399    auto * tok_embd = model.tok_embd;400    if (tok_embd == nullptr) {401        GGML_ASSERT(cparams.ctx_other != nullptr);402        const auto * model_other = llama_get_model(cparams.ctx_other);403 404        GGML_ASSERT(model_other->tok_embd != nullptr && "DFlash decoder requires the target model's token embeddings");405        tok_embd = model_other->tok_embd;406    }407 408    auto inp = std::make_unique<llm_graph_input_embd>(n_embd);409 410    inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);411    ggml_set_input(inp->tokens);412 413    ggml_tensor * inp_tokens = inp->tokens;414 415    ggml_tensor * inpL = ggml_get_rows(ctx0, tok_embd, inp->tokens);416    cb(inpL, "inp_noise_embd", -1);417 418    res->add_input(std::move(inp));419 420    for (int il = 0; il < n_layer; ++il) {421        const auto & layer = model.layers[il];422 423        ggml_tensor * noise_norm = build_norm(inpL, layer.attn_norm, NULL, LLM_NORM_RMS, il);424        cb(noise_norm, "noise_norm", il);425 426        ggml_tensor * Qcur = build_lora_mm(layer.wq, noise_norm);427        ggml_tensor * Kcur = build_lora_mm(layer.wk, noise_norm);428        ggml_tensor * Vcur = build_lora_mm(layer.wv, noise_norm);429 430        Qcur = ggml_reshape_3d(ctx0, Qcur, n_embd_head, n_head,    n_tokens);431        Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);432        Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);433 434        Qcur = build_norm(Qcur, layer.attn_q_norm, NULL, LLM_NORM_RMS, il);435        Kcur = build_norm(Kcur, layer.attn_k_norm, NULL, LLM_NORM_RMS, il);436 437        Qcur = ggml_rope_ext(438                ctx0, Qcur, inp_pos, nullptr,439                n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,440                ext_factor, attn_factor, beta_fast, beta_slow441                );442        Kcur = ggml_rope_ext(443                ctx0, Kcur, inp_pos, nullptr,444                n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,445                ext_factor, attn_factor, beta_fast, beta_slow446                );447        cb(Qcur, "Qcur", il);448        cb(Kcur, "Kcur", il);449        cb(Vcur, "Vcur", il);450 451        // cache-aware, non-causal attention452        ggml_tensor * cur = use_iswa453            ? build_attn(inp_attn_iswa, layer.wo, NULL, NULL, Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il)454            : build_attn(inp_attn,      layer.wo, NULL, NULL, Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);455 456        ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpL);457        cb(ffn_inp, "ffn_inp", il);458 459        cur = build_norm(ffn_inp, layer.ffn_norm, NULL, LLM_NORM_RMS, il);460        cb(cur, "ffn_norm", il);461 462        cur = build_ffn(cur,463                layer.ffn_up,   NULL, NULL,464                layer.ffn_gate, NULL, NULL,465                layer.ffn_down, NULL, NULL,466                NULL,467                LLM_FFN_SILU, LLM_FFN_PAR, il);468        cb(cur, "ffn_out", il);469 470        cur = ggml_add(ctx0, cur, ffn_inp);471        cb(cur, "l_out", il);472 473        inpL = cur;474    }475 476    ggml_tensor * cur = build_norm(inpL, model.output_norm, NULL, LLM_NORM_RMS, -1);477    cb(cur, "result_norm", -1);478 479    res->t_embd = cur;480 481    // lm_head from the target model (shared via ctx_other)482    auto * output = model.output;483    if (output == nullptr) {484        GGML_ASSERT(cparams.ctx_other != nullptr);485        const auto * model_other = llama_get_model(cparams.ctx_other);486        GGML_ASSERT(model_other->output != nullptr && "DFlash decoder requires the target model's output projection");487        output = model_other->output;488    }489 490    cur = build_lora_mm(output, cur);491    cb(cur, "result_output", -1);492    res->t_logits = cur;493 494    ggml_build_forward_expand(gf, cur);495 496    // DSpark: bias the draft logits with the Markov head497    if (model.dspark_markov_w1) {498        build_dspark_markov_head(*this, model, inp_tokens);499    }500}501 502// DSV4 DSpark decoder, dual-mode by batch type (see the DFlash decoder above):503//   * embd batch  -> project main_x through each stage's wkv and inject K into the ring cache504//   * token batch -> noise block through 3 full DSV4 stages (hc + MLA + MoE), markov + confidence heads505llama_model_dflash::graph_dsv4::graph_dsv4(const llama_model & model, const llm_graph_params & params) :506    llama_model_deepseek4::graph(params) {507    const int64_t n_embd_head      = hparams.n_embd_head_k();508    const int64_t n_embd_head_rope = hparams.n_rot();509    const int64_t n_embd_head_nope = n_embd_head - n_embd_head_rope;510 511    ggml_tensor * inp_pos = build_inp_pos();512 513    llm_graph_input_attn_k_iswa * inp_attn = build_attn_inp_k_iswa();514 515    // KV cache injection: fused target features from the encoder516    if (ubatch.embd) {517        auto inp = std::make_unique<llm_graph_input_embd>(n_embd);518 519        inp->embd = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_embd, n_tokens);520        ggml_set_input(inp->embd);521 522        ggml_tensor * inp_g = inp->embd;523        cb(inp_g, "inp_g_embeddings", -1);524 525        res->add_input(std::move(inp));526 527        for (int il = 0; il < n_layer; ++il) {528            const auto & layer = model.layers[il];529 530            // main-track KV: kv_norm(wkv(main_x)) with rope on the trailing dims, same531            // rope parameters as the uncompressed layers in build_attention_impl532            ggml_tensor * kv = build_lora_mm(layer.wkv, inp_g);533            kv = build_norm(kv, layer.attn_kv_norm, nullptr, LLM_NORM_RMS, il);534            kv = ggml_reshape_3d(ctx0, kv, n_embd_head, 1, n_tokens);535 536            ggml_tensor * kv_nope = ggml_view_3d(ctx0, kv, n_embd_head_nope, 1, n_tokens,537                    ggml_row_size(kv->type, n_embd_head),538                    ggml_row_size(kv->type, n_embd_head),539                    0);540            ggml_tensor * kv_pe = ggml_view_3d(ctx0, kv, n_embd_head_rope, 1, n_tokens,541                    ggml_row_size(kv->type, n_embd_head),542                    ggml_row_size(kv->type, n_embd_head),543                    ggml_row_size(kv->type, n_embd_head_nope));544            kv_pe = ggml_rope_ext(ctx0, kv_pe, inp_pos, nullptr, n_embd_head_rope, rope_type, 0,545                    freq_base, 1.0f, 0.0f, 1.0f, 0.0f, 0.0f);546            kv = ggml_concat(ctx0, kv_nope, kv_pe, 0);547            cb(kv, "kv_injected", il);548 549            if (inp_attn->self_k_rot_swa) {550                kv = llama_mul_mat_hadamard(ctx0, kv, inp_attn->self_k_rot_swa);551            }552            ggml_build_forward_expand(gf, inp_attn->mctx->get_swa()->cpy_k(ctx0, kv, inp_attn->get_k_idxs_swa(), il));553        }554 555        res->t_embd = inp_g;556 557        ggml_build_forward_expand(gf, inp_g);558        return;559    }560 561    // tok_embd from the target model (shared via ctx_other)562    auto * tok_embd = model.tok_embd;563    if (tok_embd == nullptr) {564        GGML_ASSERT(cparams.ctx_other != nullptr);565        const auto * model_other = llama_get_model(cparams.ctx_other);566 567        GGML_ASSERT(model_other->tok_embd != nullptr && "DSpark decoder requires the target model's token embeddings");568        tok_embd = model_other->tok_embd;569    }570 571    auto inp = std::make_unique<llm_graph_input_embd>(n_embd);572 573    inp->tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);574    ggml_set_input(inp->tokens);575 576    ggml_tensor * inp_tokens = inp->tokens;577 578    ggml_tensor * inpL = ggml_get_rows(ctx0, tok_embd, inp->tokens);579    cb(inpL, "inp_noise_embd", -1);580 581    res->add_input(std::move(inp));582 583    const int64_t hc = hparams.dsv4_hc_mult;584    inpL = ggml_reshape_3d(ctx0, inpL, n_embd, 1, n_tokens);585    inpL = ggml_repeat_4d(ctx0, inpL, n_embd, hc, n_tokens, 1);586    cb(inpL, "hc_init", -1);587 588    for (int il = 0; il < n_layer; ++il) {589        const auto & layer = model.layers[il];590 591        ggml_tensor * residual = inpL;592        ggml_tensor * post = nullptr;593        ggml_tensor * comb = nullptr;594 595        ggml_tensor * cur = build_hc_pre(inpL,596                layer.hc_attn_fn,597                layer.hc_attn_scale,598                layer.hc_attn_base,599                &post, &comb, il);600        cb(cur, "hc_attn_pre", il);601 602        cur = build_norm(cur, layer.attn_norm, nullptr, LLM_NORM_RMS, il);603        cb(cur, "attn_norm", il);604 605        cur = build_attention(model, inp_attn, cur, inp_pos, il);606 607        inpL = build_hc_post(cur, residual, post, comb, il);608        cb(inpL, "hc_attn_post", il);609 610        residual = inpL;611        cur = build_hc_pre(inpL,612                layer.hc_ffn_fn,613                layer.hc_ffn_scale,614                layer.hc_ffn_base,615                &post, &comb, il);616        cb(cur, "hc_ffn_pre", il);617 618        cur = build_norm(cur, layer.ffn_norm, nullptr, LLM_NORM_RMS, il);619        cb(cur, "ffn_norm", il);620 621        ggml_tensor * moe_out = build_moe_ffn(cur,622                layer.ffn_gate_inp,623                layer.ffn_up_exps,624                layer.ffn_gate_exps,625                layer.ffn_down_exps,626                layer.ffn_exp_probs_b,627                n_expert, hparams.n_expert_used,628                LLM_FFN_SILU, hparams.expert_weights_norm,629                hparams.expert_weights_scale,630                (llama_expert_gating_func_type) hparams.expert_gating_func,631                il);632        cb(moe_out, "ffn_moe_out", il);633 634        ggml_tensor * ffn_shexp = build_ffn(cur,635                layer.ffn_up_shexp, nullptr, nullptr,636                layer.ffn_gate_shexp, nullptr, nullptr,637                layer.ffn_down_shexp, nullptr, nullptr,638                nullptr, LLM_FFN_SILU, LLM_FFN_PAR, il);639        cb(ffn_shexp, "ffn_shexp", il);640 641        cur = ggml_add(ctx0, moe_out, ffn_shexp);642        cb(cur, "ffn_out", il);643 644        inpL = build_hc_post(cur, residual, post, comb, il);645        cb(inpL, "l_out", il);646    }647 648    ggml_tensor * cur = build_hc_head(inpL, model.hc_head_fn, model.hc_head_scale, model.hc_head_base);649    cb(cur, "hc_head", -1);650 651    // confidence head input: the reference scores the pre-norm collapsed hidden state652    res->t_embd = cur;653 654    cur = build_norm(cur, model.output_norm, nullptr, LLM_NORM_RMS, -1);655    cb(cur, "result_norm", -1);656 657    // lm_head from the target model (shared via ctx_other)658    auto * output = model.output;659    if (output == nullptr) {660        GGML_ASSERT(cparams.ctx_other != nullptr);661        const auto * model_other = llama_get_model(cparams.ctx_other);662        GGML_ASSERT(model_other->output != nullptr && "DSpark decoder requires the target model's output projection");663        output = model_other->output;664    }665 666    cur = build_lora_mm(output, cur);667    cb(cur, "result_output", -1);668    res->t_logits = cur;669 670    ggml_build_forward_expand(gf, cur);671 672    if (model.dspark_markov_w1) {673        build_dspark_markov_head(*this, model, inp_tokens);674    }675}676 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai