Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
granite-switch.cpp427 linesDownload Raw Back to models
1#include "models.h"2 3#include <cmath>4 5void llama_model_granite_switch::load_arch_hparams(llama_model_loader & ml) {6    ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);7    ml.get_key(LLM_KV_LOGIT_SCALE,                 hparams.f_logit_scale);8    ml.get_key(LLM_KV_RESIDUAL_SCALE,              hparams.f_residual_scale, false);9    ml.get_key(LLM_KV_EMBEDDING_SCALE,             hparams.f_embedding_scale, false);10    ml.get_key(LLM_KV_ATTENTION_SCALE,             hparams.f_attention_scale, false);11 12    bool rope_finetuned = true;13    ml.get_key(LLM_KV_ROPE_SCALING_FINETUNED, rope_finetuned, false);14    hparams.rope_finetuned = rope_finetuned;15 16    switch (hparams.n_layer()) {17        case 40: type = hparams.n_embd == 4096 ? LLM_TYPE_8B : LLM_TYPE_3B; break;18        case 64: type = LLM_TYPE_30B; break;19        default: type = LLM_TYPE_UNKNOWN;20    }21 22    ml.get_key(LLM_KV_EXPERT_SHARED_FEED_FORWARD_LENGTH, hparams.n_ff_shexp, /* required */ false);23 24    ml.get_key(LLM_KV_ADAPTER_COUNT,     n_adapters);25    ml.get_key(LLM_KV_ADAPTER_LORA_RANK, max_lora_rank);26    ml.get_key(LLM_KV_ADAPTER_ROUTER_GAIN, router_gain, /* required */ false);27 28    // bound counts that size tensors29    if (n_adapters > 4096) {30        throw std::runtime_error(format("graniteswitch: invalid adapter count %u", n_adapters));31    }32    if (max_lora_rank > 4096) {33        throw std::runtime_error(format("graniteswitch: invalid lora rank %u", max_lora_rank));34    }35 36    std::vector<llama_token> token_ids;37    std::vector<llama_token> substitute_ids;38    ml.get_arr(LLM_KV_ADAPTER_TOKEN_IDS_ACTIVATE,   token_ids);39    ml.get_arr(LLM_KV_ADAPTER_TOKEN_IDS_SUBSTITUTE, substitute_ids);40 41    if (token_ids.size() != n_adapters || substitute_ids.size() != n_adapters) {42        throw std::runtime_error(format(43            "graniteswitch: adapter token id arrays (%zu activate, %zu substitute) do not match adapter count %u",44            token_ids.size(), substitute_ids.size(), n_adapters));45    }46 47    adapter_token_to_slot.clear();48    adapter_token_to_substitute.clear();49    for (uint32_t i = 0; i < n_adapters; ++i) {50        // adapter i -> stacked slot i+1 (slot 0 is the base/zero delta)51        adapter_token_to_slot[token_ids[i]]       = (int32_t) (i + 1);52        adapter_token_to_substitute[token_ids[i]] = substitute_ids[i];53    }54 55    // extra single-head attention layer at the END (index n_real) holds the router56    // K/V. reusing n_layer_nextn keeps n_layer() == n_real, so the regular layers57    // keep their indices and the KV cache shift/defrag skips the router layer.58    // n_layer_nextn is repurposed here (no MTP): it leaks as 1 into the59    // llama_model_n_layer_nextn() getter and a re-saved nextn_predict_layers60    const uint32_t n_real = hparams.n_layer();61    if (n_real >= LLAMA_MAX_LAYERS) {62        throw std::runtime_error(format("graniteswitch: block count %u exceeds LLAMA_MAX_LAYERS", n_real));63    }64    hparams.router_layer  = (int32_t) n_real;65    hparams.n_layer_all   = n_real + 1;66    hparams.n_layer_nextn = 1;67 68    hparams.n_head_arr[n_real]    = 1;69    hparams.n_head_kv_arr[n_real] = 1;70    hparams.n_ff_arr[n_real]      = 0;71}72 73void llama_model_granite_switch::load_arch_tensors(llama_model_loader &) {74    LLAMA_LOAD_LOCALS;75 76    const int64_t n_slots     = (int64_t) n_adapters + 1; // slot 0 = base/zero delta77    const int64_t n_rank      = (int64_t) max_lora_rank;78    const int64_t n_embd_q    = n_embd_head_k * n_head;79    const int64_t n_embd_kv   = n_embd_k_gqa;80 81    tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);82 83    // substitute ids index tok_embd rows directly; range-check against n_vocab84    for (const auto & kv : adapter_token_to_substitute) {85        const llama_token sub = kv.second;86        if (sub < 0 || (int64_t) sub >= n_vocab) {87            throw std::runtime_error(format(88                "graniteswitch: substitute token id %d out of range [0, %d)", sub, (int) n_vocab));89        }90    }91 92    output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);93    output      = create_tensor(tn(LLM_TENSOR_OUTPUT,      "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED);94    if (output == NULL) {95        output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);96    }97 98    for (int i = 0; i < n_layer; ++i) {99        auto & layer = layers[i];100 101        layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);102 103        layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", i), {n_embd, n_embd_q + 2*n_embd_kv}, 0);104        layer.wo   = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd_q, n_embd}, 0);105 106        layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);107 108        layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd,   n_ff}, 0);109        layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), {  n_ff, n_embd}, 0);110        layer.ffn_up   = create_tensor(tn(LLM_TENSOR_FFN_UP,   "weight", i), {n_embd,   n_ff}, 0);111 112        auto & sl = layer.switch_lora;113 114        sl.a_q = create_tensor(tn(LLM_TENSOR_ATTN_Q, "lora_a", i), {n_embd,  n_rank, n_slots}, 0);115        sl.b_q = create_tensor(tn(LLM_TENSOR_ATTN_Q, "lora_b", i), {n_rank, n_embd_q, n_slots}, 0);116        sl.a_k = create_tensor(tn(LLM_TENSOR_ATTN_K, "lora_a", i), {n_embd,  n_rank, n_slots}, 0);117        sl.b_k = create_tensor(tn(LLM_TENSOR_ATTN_K, "lora_b", i), {n_rank, n_embd_kv, n_slots}, 0);118        sl.a_v = create_tensor(tn(LLM_TENSOR_ATTN_V, "lora_a", i), {n_embd,  n_rank, n_slots}, 0);119        sl.b_v = create_tensor(tn(LLM_TENSOR_ATTN_V, "lora_b", i), {n_rank, n_embd_kv, n_slots}, 0);120 121        sl.a_o = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "lora_a", i), {n_embd_q, n_rank, n_slots}, 0);122        sl.b_o = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "lora_b", i), {n_rank,   n_embd, n_slots}, 0);123 124        sl.a_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "lora_a", i), {n_embd, n_rank, n_slots}, 0);125        sl.b_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "lora_b", i), {n_rank,  n_ff,  n_slots}, 0);126        sl.a_up   = create_tensor(tn(LLM_TENSOR_FFN_UP,   "lora_a", i), {n_embd, n_rank, n_slots}, 0);127        sl.b_up   = create_tensor(tn(LLM_TENSOR_FFN_UP,   "lora_b", i), {n_rank,  n_ff,  n_slots}, 0);128        sl.a_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "lora_a", i), {  n_ff, n_rank, n_slots}, 0);129        sl.b_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "lora_b", i), {n_rank, n_embd, n_slots}, 0);130    }131}132 133class llm_graph_input_switch : public llm_graph_input_i {134public:135    llm_graph_input_switch(const llama_model_granite_switch & smodel) : smodel(smodel) {}136    virtual ~llm_graph_input_switch() = default;137 138    void set_input(const llama_ubatch * ubatch) override;139 140    ggml_tensor * sub_tokens  = nullptr; // I32 [n_tokens] adapter-substituted token ids141    ggml_tensor * router_ksig = nullptr; // F32 [n_tokens] router K signal (+/-gain)142    ggml_tensor * router_vval = nullptr; // F32 [n_tokens] router V value (adapter slot / 0)143    ggml_tensor * router_q    = nullptr; // F32 [n_tokens] router Q value (constant 1.0)144 145    const llama_model_granite_switch & smodel;146};147 148// K dim-0 is +gain for an adapter token, -gain otherwise; the causal softmax then149// lets a single visible adapter token dominate so the readback recovers its slot.150void llm_graph_input_switch::set_input(const llama_ubatch * ubatch) {151    if (!ubatch->token) {152        return;153    }154 155    const int64_t n_tokens = ubatch->n_tokens;156 157    std::vector<int32_t> sub (n_tokens);158    std::vector<float>   ksig(n_tokens);159    std::vector<float>   vval(n_tokens);160    std::vector<float>   q   (n_tokens, 1.0f);161 162    for (int64_t i = 0; i < n_tokens; ++i) {163        const llama_token tok = ubatch->token[i];164 165        const auto it = smodel.adapter_token_to_slot.find(tok);166        if (it != smodel.adapter_token_to_slot.end()) {167            ksig[i] = +smodel.router_gain;168            vval[i] = (float) it->second;169        } else {170            ksig[i] = -smodel.router_gain;171            vval[i] = 0.0f;172        }173 174        const auto sit = smodel.adapter_token_to_substitute.find(tok);175        sub[i] = (sit != smodel.adapter_token_to_substitute.end())176            ? (int32_t) sit->second177            : (int32_t) tok;178    }179 180    ggml_backend_tensor_set(sub_tokens,  sub.data(),  0, n_tokens*ggml_element_size(sub_tokens));181    ggml_backend_tensor_set(router_ksig, ksig.data(), 0, n_tokens*ggml_element_size(router_ksig));182    ggml_backend_tensor_set(router_vval, vval.data(), 0, n_tokens*ggml_element_size(router_vval));183    ggml_backend_tensor_set(router_q,    q.data(),    0, n_tokens*ggml_element_size(router_q));184}185 186std::unique_ptr<llm_graph_context> llama_model_granite_switch::build_arch_graph(const llm_graph_params & params) const {187    return std::make_unique<graph>(*this, params);188}189 190// per-token switched LoRA delta: B_a*(A_a*x), adapter selected per token via ids.191// cur: {n_in, n_tokens}, ids: {n_tokens} -> {n_out, n_tokens}192ggml_tensor * llama_model_granite_switch::graph::build_switched_lora_delta(193          ggml_tensor * lora_a,194          ggml_tensor * lora_b,195          ggml_tensor * cur,196          ggml_tensor * ids) {197    const int64_t n_in     = cur->ne[0];198    const int64_t n_tokens = cur->ne[1];199 200    ggml_tensor * x    = ggml_reshape_3d(ctx0, cur, n_in, 1, n_tokens);201    ggml_tensor * ids2 = ggml_reshape_2d(ctx0, ids, 1, n_tokens);202 203    ggml_tensor * a = ggml_mul_mat_id(ctx0, lora_a, x, ids2); // {max_rank, 1, n_tokens}204    ggml_tensor * d = ggml_mul_mat_id(ctx0, lora_b, a, ids2); // {n_out,    1, n_tokens}205 206    return ggml_reshape_2d(ctx0, d, d->ne[0], n_tokens);207}208 209ggml_tensor * llama_model_granite_switch::graph::build_switched_lora_mm(210          ggml_tensor * w,211          ggml_tensor * lora_a,212          ggml_tensor * lora_b,213          ggml_tensor * cur,214          ggml_tensor * ids) {215    ggml_tensor * base  = ggml_mul_mat(ctx0, w, cur);216    ggml_tensor * delta = build_switched_lora_delta(lora_a, lora_b, cur, ids);217    return ggml_add(ctx0, base, delta);218}219 220llama_model_granite_switch::graph::graph(221    const llama_model & model,222    const llm_graph_params & params)223    : llm_graph_context(params) {224 225    const auto & smodel = static_cast<const llama_model_granite_switch &>(model);226 227    // TODO: support raw embedding input (multimodal / pre-embedded tokens) when needed228    GGML_ASSERT(ubatch.token && "granite-switch requires token input");229 230    const int64_t n_embd_head = hparams.n_embd_head_v();231    GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());232    GGML_ASSERT(n_embd_head == n_rot);233 234    auto inp_switch = std::make_unique<llm_graph_input_switch>(smodel);235    inp_switch->sub_tokens  = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);236    inp_switch->router_ksig = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_tokens);237    inp_switch->router_vval = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_tokens);238    inp_switch->router_q    = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_tokens);239    ggml_set_input(inp_switch->sub_tokens);240    ggml_set_input(inp_switch->router_ksig);241    ggml_set_input(inp_switch->router_vval);242    ggml_set_input(inp_switch->router_q);243    ggml_tensor * sub_tokens  = inp_switch->sub_tokens;244    ggml_tensor * router_ksig = inp_switch->router_ksig;245    ggml_tensor * router_vval = inp_switch->router_vval;246    ggml_tensor * router_q    = inp_switch->router_q;247    res->add_input(std::move(inp_switch));248 249    // embed the substituted ids directly; build_inp_embd would embed the raw tokens250    ggml_tensor * inpL = ggml_get_rows(ctx0, model.tok_embd, sub_tokens);251    if (hparams.f_embedding_scale != 0.0f) {252        inpL = ggml_scale(ctx0, inpL, hparams.f_embedding_scale);253    }254    cb(inpL, "inp_embd", -1);255 256    ggml_tensor * inp_pos = nullptr;257    if (hparams.rope_finetuned) {258        inp_pos = build_inp_pos();259    }260    auto * inp_attn = build_attn_inp_kv();261 262    // single causal head at layer R recovers the adapter index in-graph: only dim 0263    // carries signal (Q[0]=1, K[0]=+/-gain, V[0]=slot/0), the rest is zero-padded.264    const int R = hparams.router_layer;265    GGML_ASSERT(R >= 0);266    auto router_lane = [&](ggml_tensor * sig1d) {267        ggml_tensor * t = ggml_reshape_3d(ctx0, sig1d, 1, 1, n_tokens);268        return ggml_pad(ctx0, t, (int) n_embd_head - 1, 0, 0, 0);269    };270    ggml_tensor * Qr = router_lane(router_q);271    ggml_tensor * Kr = router_lane(router_ksig);272    ggml_tensor * Vr = router_lane(router_vval);273 274    ggml_tensor * router_out = build_attn(inp_attn,275            nullptr, nullptr, nullptr,276            Qr, Kr, Vr, nullptr, nullptr, nullptr, /*kq_scale=*/1.0f, /*il=*/R);277    cb(router_out, "router_out", R);278 279    // row 0 of router_out is the attended slot; clamp+round to an I32 index280    ggml_tensor * slot_f = ggml_cont(ctx0,281        ggml_view_2d(ctx0, router_out, 1, n_tokens, router_out->nb[1], 0));282    slot_f = ggml_reshape_1d(ctx0, slot_f, n_tokens);283    slot_f = ggml_clamp(ctx0, slot_f, 0.0f, (float) smodel.n_adapters);284    slot_f = ggml_round(ctx0, slot_f);285    ggml_tensor * adapter_ids = ggml_cast(ctx0, slot_f, GGML_TYPE_I32);286    cb(adapter_ids, "adapter_ids", -1);287 288    ggml_tensor * inp_out_ids = build_inp_out_ids();289 290    ggml_tensor * cur;291 292    for (int il = 0; il < n_layer; ++il) {293        ggml_tensor * inpSA = inpL;294 295        cur = build_norm(inpL, model.layers[il].attn_norm, NULL, LLM_NORM_RMS, il);296        cb(cur, "attn_norm", il);297 298        cur = build_attention_layer(cur, inp_pos, adapter_ids, inp_attn, model, n_embd_head, il);299 300        if (il == n_layer - 1 && inp_out_ids) {301            cur   = ggml_get_rows(ctx0, cur,   inp_out_ids);302            inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids);303            // keep adapter_ids aligned to the kept rows (2D round-trip for get_rows)304            const int64_t n_out = inp_out_ids->ne[0];305            adapter_ids = ggml_get_rows(ctx0,306                ggml_reshape_2d(ctx0, adapter_ids, 1, adapter_ids->ne[0]), inp_out_ids);307            adapter_ids = ggml_reshape_1d(ctx0, adapter_ids, n_out);308        }309 310        cur = build_layer_ffn(cur, inpSA, adapter_ids, model, il);311 312        inpL = cur;313    }314 315    cur = inpL;316 317    cur = build_norm(cur, model.output_norm, NULL, LLM_NORM_RMS, -1);318    cb(cur, "result_norm", -1);319    res->t_embd = cur;320 321    cur = build_lora_mm(model.output, cur, model.output_s);322 323    cur = ggml_scale(ctx0, cur, 1.0f / hparams.f_logit_scale);324    cb(cur, "result_output", -1);325    res->t_logits = cur;326 327    ggml_build_forward_expand(gf, cur);328}329 330ggml_tensor * llama_model_granite_switch::graph::build_attention_layer(331          ggml_tensor             * cur,332          ggml_tensor             * inp_pos,333          ggml_tensor             * adapter_ids,334          llm_graph_input_attn_kv * inp_attn,335    const llama_model             & model,336    const int64_t                 n_embd_head,337    const int                     il) {338 339    const auto & layer = model.layers[il];340    const auto & sl    = layer.switch_lora;341 342    const int64_t n_head    = hparams.n_head(il);343    const int64_t n_head_kv = hparams.n_head_kv(il);344 345    ggml_tensor * qkv = ggml_mul_mat(ctx0, layer.wqkv, cur);346    cb(qkv, "wqkv", il);347 348    const int64_t n_embd_q  = n_embd_head * n_head;349    const int64_t n_embd_kv = n_embd_head * n_head_kv;350 351    // slice fused qkv into Q/K/V, made contiguous so LoRA deltas can be added352    ggml_tensor * Qcur = ggml_cont(ctx0, ggml_view_2d(ctx0, qkv, n_embd_q,  qkv->ne[1], qkv->nb[1], 0));353    ggml_tensor * Kcur = ggml_cont(ctx0, ggml_view_2d(ctx0, qkv, n_embd_kv, qkv->ne[1], qkv->nb[1], n_embd_q*ggml_element_size(qkv)));354    ggml_tensor * Vcur = ggml_cont(ctx0, ggml_view_2d(ctx0, qkv, n_embd_kv, qkv->ne[1], qkv->nb[1], (n_embd_q + n_embd_kv)*ggml_element_size(qkv)));355 356    Qcur = ggml_add(ctx0, Qcur, build_switched_lora_delta(sl.a_q, sl.b_q, cur, adapter_ids));357    Kcur = ggml_add(ctx0, Kcur, build_switched_lora_delta(sl.a_k, sl.b_k, cur, adapter_ids));358    Vcur = ggml_add(ctx0, Vcur, build_switched_lora_delta(sl.a_v, sl.b_v, cur, adapter_ids));359 360    Qcur = ggml_reshape_3d(ctx0, Qcur, n_embd_head, n_head,    n_tokens);361    Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);362    Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);363 364    if (hparams.rope_finetuned) {365        ggml_tensor * rope_factors = model.get_rope_factors(cparams, il);366        Qcur = ggml_rope_ext(ctx0, Qcur, inp_pos, rope_factors,367                n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,368                ext_factor, attn_factor, beta_fast, beta_slow);369        Kcur = ggml_rope_ext(ctx0, Kcur, inp_pos, rope_factors,370                n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,371                ext_factor, attn_factor, beta_fast, beta_slow);372    }373    cb(Qcur, "Qcur", il);374    cb(Kcur, "Kcur", il);375    cb(Vcur, "Vcur", il);376 377    const float kq_scale = hparams.f_attention_scale == 0.0f378        ? 1.0f/sqrtf(float(n_embd_head)) : hparams.f_attention_scale;379 380    // wo = nullptr so build_attn returns concatenated heads; o-proj is switched below381    ggml_tensor * attn = build_attn(inp_attn,382            nullptr, nullptr, nullptr,383            Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);384    cb(attn, "attn_pre_o", il);385 386    cur = build_switched_lora_mm(layer.wo, sl.a_o, sl.b_o, attn, adapter_ids);387    cb(cur, "attn_out", il);388    return cur;389}390 391ggml_tensor * llama_model_granite_switch::graph::build_layer_ffn(392          ggml_tensor       * cur,393          ggml_tensor       * inpSA,394          ggml_tensor       * adapter_ids,395    const llama_model       & model,396    const int                 il) {397 398    const auto & layer = model.layers[il];399    const auto & sl    = layer.switch_lora;400 401    if (hparams.f_residual_scale) {402        cur = ggml_scale(ctx0, cur, hparams.f_residual_scale);403    }404    ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA);405    cb(ffn_inp, "ffn_inp", il);406 407    cur = build_norm(ffn_inp, layer.ffn_norm, NULL, LLM_NORM_RMS, il);408    cb(cur, "ffn_norm", il);409 410    ggml_tensor * g = build_switched_lora_mm(layer.ffn_gate, sl.a_gate, sl.b_gate, cur, adapter_ids);411    ggml_tensor * u = build_switched_lora_mm(layer.ffn_up,   sl.a_up,   sl.b_up,   cur, adapter_ids);412    g = ggml_silu(ctx0, g);413    ggml_tensor * gu = ggml_mul(ctx0, g, u);414    cur = build_switched_lora_mm(layer.ffn_down, sl.a_down, sl.b_down, gu, adapter_ids);415    cb(cur, "ffn_out", il);416 417    if (hparams.f_residual_scale) {418        cur = ggml_scale(ctx0, cur, hparams.f_residual_scale);419    }420    cur = ggml_add(ctx0, cur, ffn_inp);421 422    cur = build_cvec(cur, il);423    cb(cur, "l_out", il);424 425    return cur;426}427 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai