Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#include "models.h"2 3#include <cmath>4 5void llama_model_granite_switch::load_arch_hparams(llama_model_loader & ml) {6 ml.get_key(LLM_KV_ATTENTION_LAYERNORM_RMS_EPS, hparams.f_norm_rms_eps);7 ml.get_key(LLM_KV_LOGIT_SCALE, hparams.f_logit_scale);8 ml.get_key(LLM_KV_RESIDUAL_SCALE, hparams.f_residual_scale, false);9 ml.get_key(LLM_KV_EMBEDDING_SCALE, hparams.f_embedding_scale, false);10 ml.get_key(LLM_KV_ATTENTION_SCALE, hparams.f_attention_scale, false);11 12 bool rope_finetuned = true;13 ml.get_key(LLM_KV_ROPE_SCALING_FINETUNED, rope_finetuned, false);14 hparams.rope_finetuned = rope_finetuned;15 16 switch (hparams.n_layer()) {17 case 40: type = hparams.n_embd == 4096 ? LLM_TYPE_8B : LLM_TYPE_3B; break;18 case 64: type = LLM_TYPE_30B; break;19 default: type = LLM_TYPE_UNKNOWN;20 }21 22 ml.get_key(LLM_KV_EXPERT_SHARED_FEED_FORWARD_LENGTH, hparams.n_ff_shexp, /* required */ false);23 24 ml.get_key(LLM_KV_ADAPTER_COUNT, n_adapters);25 ml.get_key(LLM_KV_ADAPTER_LORA_RANK, max_lora_rank);26 ml.get_key(LLM_KV_ADAPTER_ROUTER_GAIN, router_gain, /* required */ false);27 28 // bound counts that size tensors29 if (n_adapters > 4096) {30 throw std::runtime_error(format("graniteswitch: invalid adapter count %u", n_adapters));31 }32 if (max_lora_rank > 4096) {33 throw std::runtime_error(format("graniteswitch: invalid lora rank %u", max_lora_rank));34 }35 36 std::vector<llama_token> token_ids;37 std::vector<llama_token> substitute_ids;38 ml.get_arr(LLM_KV_ADAPTER_TOKEN_IDS_ACTIVATE, token_ids);39 ml.get_arr(LLM_KV_ADAPTER_TOKEN_IDS_SUBSTITUTE, substitute_ids);40 41 if (token_ids.size() != n_adapters || substitute_ids.size() != n_adapters) {42 throw std::runtime_error(format(43 "graniteswitch: adapter token id arrays (%zu activate, %zu substitute) do not match adapter count %u",44 token_ids.size(), substitute_ids.size(), n_adapters));45 }46 47 adapter_token_to_slot.clear();48 adapter_token_to_substitute.clear();49 for (uint32_t i = 0; i < n_adapters; ++i) {50 // adapter i -> stacked slot i+1 (slot 0 is the base/zero delta)51 adapter_token_to_slot[token_ids[i]] = (int32_t) (i + 1);52 adapter_token_to_substitute[token_ids[i]] = substitute_ids[i];53 }54 55 // extra single-head attention layer at the END (index n_real) holds the router56 // K/V. reusing n_layer_nextn keeps n_layer() == n_real, so the regular layers57 // keep their indices and the KV cache shift/defrag skips the router layer.58 // n_layer_nextn is repurposed here (no MTP): it leaks as 1 into the59 // llama_model_n_layer_nextn() getter and a re-saved nextn_predict_layers60 const uint32_t n_real = hparams.n_layer();61 if (n_real >= LLAMA_MAX_LAYERS) {62 throw std::runtime_error(format("graniteswitch: block count %u exceeds LLAMA_MAX_LAYERS", n_real));63 }64 hparams.router_layer = (int32_t) n_real;65 hparams.n_layer_all = n_real + 1;66 hparams.n_layer_nextn = 1;67 68 hparams.n_head_arr[n_real] = 1;69 hparams.n_head_kv_arr[n_real] = 1;70 hparams.n_ff_arr[n_real] = 0;71}72 73void llama_model_granite_switch::load_arch_tensors(llama_model_loader &) {74 LLAMA_LOAD_LOCALS;75 76 const int64_t n_slots = (int64_t) n_adapters + 1; // slot 0 = base/zero delta77 const int64_t n_rank = (int64_t) max_lora_rank;78 const int64_t n_embd_q = n_embd_head_k * n_head;79 const int64_t n_embd_kv = n_embd_k_gqa;80 81 tok_embd = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, 0);82 83 // substitute ids index tok_embd rows directly; range-check against n_vocab84 for (const auto & kv : adapter_token_to_substitute) {85 const llama_token sub = kv.second;86 if (sub < 0 || (int64_t) sub >= n_vocab) {87 throw std::runtime_error(format(88 "graniteswitch: substitute token id %d out of range [0, %d)", sub, (int) n_vocab));89 }90 }91 92 output_norm = create_tensor(tn(LLM_TENSOR_OUTPUT_NORM, "weight"), {n_embd}, 0);93 output = create_tensor(tn(LLM_TENSOR_OUTPUT, "weight"), {n_embd, n_vocab}, TENSOR_NOT_REQUIRED);94 if (output == NULL) {95 output = create_tensor(tn(LLM_TENSOR_TOKEN_EMBD, "weight"), {n_embd, n_vocab}, TENSOR_DUPLICATED);96 }97 98 for (int i = 0; i < n_layer; ++i) {99 auto & layer = layers[i];100 101 layer.attn_norm = create_tensor(tn(LLM_TENSOR_ATTN_NORM, "weight", i), {n_embd}, 0);102 103 layer.wqkv = create_tensor(tn(LLM_TENSOR_ATTN_QKV, "weight", i), {n_embd, n_embd_q + 2*n_embd_kv}, 0);104 layer.wo = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "weight", i), {n_embd_q, n_embd}, 0);105 106 layer.ffn_norm = create_tensor(tn(LLM_TENSOR_FFN_NORM, "weight", i), {n_embd}, 0);107 108 layer.ffn_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "weight", i), {n_embd, n_ff}, 0);109 layer.ffn_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "weight", i), { n_ff, n_embd}, 0);110 layer.ffn_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "weight", i), {n_embd, n_ff}, 0);111 112 auto & sl = layer.switch_lora;113 114 sl.a_q = create_tensor(tn(LLM_TENSOR_ATTN_Q, "lora_a", i), {n_embd, n_rank, n_slots}, 0);115 sl.b_q = create_tensor(tn(LLM_TENSOR_ATTN_Q, "lora_b", i), {n_rank, n_embd_q, n_slots}, 0);116 sl.a_k = create_tensor(tn(LLM_TENSOR_ATTN_K, "lora_a", i), {n_embd, n_rank, n_slots}, 0);117 sl.b_k = create_tensor(tn(LLM_TENSOR_ATTN_K, "lora_b", i), {n_rank, n_embd_kv, n_slots}, 0);118 sl.a_v = create_tensor(tn(LLM_TENSOR_ATTN_V, "lora_a", i), {n_embd, n_rank, n_slots}, 0);119 sl.b_v = create_tensor(tn(LLM_TENSOR_ATTN_V, "lora_b", i), {n_rank, n_embd_kv, n_slots}, 0);120 121 sl.a_o = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "lora_a", i), {n_embd_q, n_rank, n_slots}, 0);122 sl.b_o = create_tensor(tn(LLM_TENSOR_ATTN_OUT, "lora_b", i), {n_rank, n_embd, n_slots}, 0);123 124 sl.a_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "lora_a", i), {n_embd, n_rank, n_slots}, 0);125 sl.b_gate = create_tensor(tn(LLM_TENSOR_FFN_GATE, "lora_b", i), {n_rank, n_ff, n_slots}, 0);126 sl.a_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "lora_a", i), {n_embd, n_rank, n_slots}, 0);127 sl.b_up = create_tensor(tn(LLM_TENSOR_FFN_UP, "lora_b", i), {n_rank, n_ff, n_slots}, 0);128 sl.a_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "lora_a", i), { n_ff, n_rank, n_slots}, 0);129 sl.b_down = create_tensor(tn(LLM_TENSOR_FFN_DOWN, "lora_b", i), {n_rank, n_embd, n_slots}, 0);130 }131}132 133class llm_graph_input_switch : public llm_graph_input_i {134public:135 llm_graph_input_switch(const llama_model_granite_switch & smodel) : smodel(smodel) {}136 virtual ~llm_graph_input_switch() = default;137 138 void set_input(const llama_ubatch * ubatch) override;139 140 ggml_tensor * sub_tokens = nullptr; // I32 [n_tokens] adapter-substituted token ids141 ggml_tensor * router_ksig = nullptr; // F32 [n_tokens] router K signal (+/-gain)142 ggml_tensor * router_vval = nullptr; // F32 [n_tokens] router V value (adapter slot / 0)143 ggml_tensor * router_q = nullptr; // F32 [n_tokens] router Q value (constant 1.0)144 145 const llama_model_granite_switch & smodel;146};147 148// K dim-0 is +gain for an adapter token, -gain otherwise; the causal softmax then149// lets a single visible adapter token dominate so the readback recovers its slot.150void llm_graph_input_switch::set_input(const llama_ubatch * ubatch) {151 if (!ubatch->token) {152 return;153 }154 155 const int64_t n_tokens = ubatch->n_tokens;156 157 std::vector<int32_t> sub (n_tokens);158 std::vector<float> ksig(n_tokens);159 std::vector<float> vval(n_tokens);160 std::vector<float> q (n_tokens, 1.0f);161 162 for (int64_t i = 0; i < n_tokens; ++i) {163 const llama_token tok = ubatch->token[i];164 165 const auto it = smodel.adapter_token_to_slot.find(tok);166 if (it != smodel.adapter_token_to_slot.end()) {167 ksig[i] = +smodel.router_gain;168 vval[i] = (float) it->second;169 } else {170 ksig[i] = -smodel.router_gain;171 vval[i] = 0.0f;172 }173 174 const auto sit = smodel.adapter_token_to_substitute.find(tok);175 sub[i] = (sit != smodel.adapter_token_to_substitute.end())176 ? (int32_t) sit->second177 : (int32_t) tok;178 }179 180 ggml_backend_tensor_set(sub_tokens, sub.data(), 0, n_tokens*ggml_element_size(sub_tokens));181 ggml_backend_tensor_set(router_ksig, ksig.data(), 0, n_tokens*ggml_element_size(router_ksig));182 ggml_backend_tensor_set(router_vval, vval.data(), 0, n_tokens*ggml_element_size(router_vval));183 ggml_backend_tensor_set(router_q, q.data(), 0, n_tokens*ggml_element_size(router_q));184}185 186std::unique_ptr<llm_graph_context> llama_model_granite_switch::build_arch_graph(const llm_graph_params & params) const {187 return std::make_unique<graph>(*this, params);188}189 190// per-token switched LoRA delta: B_a*(A_a*x), adapter selected per token via ids.191// cur: {n_in, n_tokens}, ids: {n_tokens} -> {n_out, n_tokens}192ggml_tensor * llama_model_granite_switch::graph::build_switched_lora_delta(193 ggml_tensor * lora_a,194 ggml_tensor * lora_b,195 ggml_tensor * cur,196 ggml_tensor * ids) {197 const int64_t n_in = cur->ne[0];198 const int64_t n_tokens = cur->ne[1];199 200 ggml_tensor * x = ggml_reshape_3d(ctx0, cur, n_in, 1, n_tokens);201 ggml_tensor * ids2 = ggml_reshape_2d(ctx0, ids, 1, n_tokens);202 203 ggml_tensor * a = ggml_mul_mat_id(ctx0, lora_a, x, ids2); // {max_rank, 1, n_tokens}204 ggml_tensor * d = ggml_mul_mat_id(ctx0, lora_b, a, ids2); // {n_out, 1, n_tokens}205 206 return ggml_reshape_2d(ctx0, d, d->ne[0], n_tokens);207}208 209ggml_tensor * llama_model_granite_switch::graph::build_switched_lora_mm(210 ggml_tensor * w,211 ggml_tensor * lora_a,212 ggml_tensor * lora_b,213 ggml_tensor * cur,214 ggml_tensor * ids) {215 ggml_tensor * base = ggml_mul_mat(ctx0, w, cur);216 ggml_tensor * delta = build_switched_lora_delta(lora_a, lora_b, cur, ids);217 return ggml_add(ctx0, base, delta);218}219 220llama_model_granite_switch::graph::graph(221 const llama_model & model,222 const llm_graph_params & params)223 : llm_graph_context(params) {224 225 const auto & smodel = static_cast<const llama_model_granite_switch &>(model);226 227 // TODO: support raw embedding input (multimodal / pre-embedded tokens) when needed228 GGML_ASSERT(ubatch.token && "granite-switch requires token input");229 230 const int64_t n_embd_head = hparams.n_embd_head_v();231 GGML_ASSERT(n_embd_head == hparams.n_embd_head_k());232 GGML_ASSERT(n_embd_head == n_rot);233 234 auto inp_switch = std::make_unique<llm_graph_input_switch>(smodel);235 inp_switch->sub_tokens = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_tokens);236 inp_switch->router_ksig = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_tokens);237 inp_switch->router_vval = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_tokens);238 inp_switch->router_q = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_tokens);239 ggml_set_input(inp_switch->sub_tokens);240 ggml_set_input(inp_switch->router_ksig);241 ggml_set_input(inp_switch->router_vval);242 ggml_set_input(inp_switch->router_q);243 ggml_tensor * sub_tokens = inp_switch->sub_tokens;244 ggml_tensor * router_ksig = inp_switch->router_ksig;245 ggml_tensor * router_vval = inp_switch->router_vval;246 ggml_tensor * router_q = inp_switch->router_q;247 res->add_input(std::move(inp_switch));248 249 // embed the substituted ids directly; build_inp_embd would embed the raw tokens250 ggml_tensor * inpL = ggml_get_rows(ctx0, model.tok_embd, sub_tokens);251 if (hparams.f_embedding_scale != 0.0f) {252 inpL = ggml_scale(ctx0, inpL, hparams.f_embedding_scale);253 }254 cb(inpL, "inp_embd", -1);255 256 ggml_tensor * inp_pos = nullptr;257 if (hparams.rope_finetuned) {258 inp_pos = build_inp_pos();259 }260 auto * inp_attn = build_attn_inp_kv();261 262 // single causal head at layer R recovers the adapter index in-graph: only dim 0263 // carries signal (Q[0]=1, K[0]=+/-gain, V[0]=slot/0), the rest is zero-padded.264 const int R = hparams.router_layer;265 GGML_ASSERT(R >= 0);266 auto router_lane = [&](ggml_tensor * sig1d) {267 ggml_tensor * t = ggml_reshape_3d(ctx0, sig1d, 1, 1, n_tokens);268 return ggml_pad(ctx0, t, (int) n_embd_head - 1, 0, 0, 0);269 };270 ggml_tensor * Qr = router_lane(router_q);271 ggml_tensor * Kr = router_lane(router_ksig);272 ggml_tensor * Vr = router_lane(router_vval);273 274 ggml_tensor * router_out = build_attn(inp_attn,275 nullptr, nullptr, nullptr,276 Qr, Kr, Vr, nullptr, nullptr, nullptr, /*kq_scale=*/1.0f, /*il=*/R);277 cb(router_out, "router_out", R);278 279 // row 0 of router_out is the attended slot; clamp+round to an I32 index280 ggml_tensor * slot_f = ggml_cont(ctx0,281 ggml_view_2d(ctx0, router_out, 1, n_tokens, router_out->nb[1], 0));282 slot_f = ggml_reshape_1d(ctx0, slot_f, n_tokens);283 slot_f = ggml_clamp(ctx0, slot_f, 0.0f, (float) smodel.n_adapters);284 slot_f = ggml_round(ctx0, slot_f);285 ggml_tensor * adapter_ids = ggml_cast(ctx0, slot_f, GGML_TYPE_I32);286 cb(adapter_ids, "adapter_ids", -1);287 288 ggml_tensor * inp_out_ids = build_inp_out_ids();289 290 ggml_tensor * cur;291 292 for (int il = 0; il < n_layer; ++il) {293 ggml_tensor * inpSA = inpL;294 295 cur = build_norm(inpL, model.layers[il].attn_norm, NULL, LLM_NORM_RMS, il);296 cb(cur, "attn_norm", il);297 298 cur = build_attention_layer(cur, inp_pos, adapter_ids, inp_attn, model, n_embd_head, il);299 300 if (il == n_layer - 1 && inp_out_ids) {301 cur = ggml_get_rows(ctx0, cur, inp_out_ids);302 inpSA = ggml_get_rows(ctx0, inpSA, inp_out_ids);303 // keep adapter_ids aligned to the kept rows (2D round-trip for get_rows)304 const int64_t n_out = inp_out_ids->ne[0];305 adapter_ids = ggml_get_rows(ctx0,306 ggml_reshape_2d(ctx0, adapter_ids, 1, adapter_ids->ne[0]), inp_out_ids);307 adapter_ids = ggml_reshape_1d(ctx0, adapter_ids, n_out);308 }309 310 cur = build_layer_ffn(cur, inpSA, adapter_ids, model, il);311 312 inpL = cur;313 }314 315 cur = inpL;316 317 cur = build_norm(cur, model.output_norm, NULL, LLM_NORM_RMS, -1);318 cb(cur, "result_norm", -1);319 res->t_embd = cur;320 321 cur = build_lora_mm(model.output, cur, model.output_s);322 323 cur = ggml_scale(ctx0, cur, 1.0f / hparams.f_logit_scale);324 cb(cur, "result_output", -1);325 res->t_logits = cur;326 327 ggml_build_forward_expand(gf, cur);328}329 330ggml_tensor * llama_model_granite_switch::graph::build_attention_layer(331 ggml_tensor * cur,332 ggml_tensor * inp_pos,333 ggml_tensor * adapter_ids,334 llm_graph_input_attn_kv * inp_attn,335 const llama_model & model,336 const int64_t n_embd_head,337 const int il) {338 339 const auto & layer = model.layers[il];340 const auto & sl = layer.switch_lora;341 342 const int64_t n_head = hparams.n_head(il);343 const int64_t n_head_kv = hparams.n_head_kv(il);344 345 ggml_tensor * qkv = ggml_mul_mat(ctx0, layer.wqkv, cur);346 cb(qkv, "wqkv", il);347 348 const int64_t n_embd_q = n_embd_head * n_head;349 const int64_t n_embd_kv = n_embd_head * n_head_kv;350 351 // slice fused qkv into Q/K/V, made contiguous so LoRA deltas can be added352 ggml_tensor * Qcur = ggml_cont(ctx0, ggml_view_2d(ctx0, qkv, n_embd_q, qkv->ne[1], qkv->nb[1], 0));353 ggml_tensor * Kcur = ggml_cont(ctx0, ggml_view_2d(ctx0, qkv, n_embd_kv, qkv->ne[1], qkv->nb[1], n_embd_q*ggml_element_size(qkv)));354 ggml_tensor * Vcur = ggml_cont(ctx0, ggml_view_2d(ctx0, qkv, n_embd_kv, qkv->ne[1], qkv->nb[1], (n_embd_q + n_embd_kv)*ggml_element_size(qkv)));355 356 Qcur = ggml_add(ctx0, Qcur, build_switched_lora_delta(sl.a_q, sl.b_q, cur, adapter_ids));357 Kcur = ggml_add(ctx0, Kcur, build_switched_lora_delta(sl.a_k, sl.b_k, cur, adapter_ids));358 Vcur = ggml_add(ctx0, Vcur, build_switched_lora_delta(sl.a_v, sl.b_v, cur, adapter_ids));359 360 Qcur = ggml_reshape_3d(ctx0, Qcur, n_embd_head, n_head, n_tokens);361 Kcur = ggml_reshape_3d(ctx0, Kcur, n_embd_head, n_head_kv, n_tokens);362 Vcur = ggml_reshape_3d(ctx0, Vcur, n_embd_head, n_head_kv, n_tokens);363 364 if (hparams.rope_finetuned) {365 ggml_tensor * rope_factors = model.get_rope_factors(cparams, il);366 Qcur = ggml_rope_ext(ctx0, Qcur, inp_pos, rope_factors,367 n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,368 ext_factor, attn_factor, beta_fast, beta_slow);369 Kcur = ggml_rope_ext(ctx0, Kcur, inp_pos, rope_factors,370 n_rot, rope_type, n_ctx_orig, freq_base, freq_scale,371 ext_factor, attn_factor, beta_fast, beta_slow);372 }373 cb(Qcur, "Qcur", il);374 cb(Kcur, "Kcur", il);375 cb(Vcur, "Vcur", il);376 377 const float kq_scale = hparams.f_attention_scale == 0.0f378 ? 1.0f/sqrtf(float(n_embd_head)) : hparams.f_attention_scale;379 380 // wo = nullptr so build_attn returns concatenated heads; o-proj is switched below381 ggml_tensor * attn = build_attn(inp_attn,382 nullptr, nullptr, nullptr,383 Qcur, Kcur, Vcur, nullptr, nullptr, nullptr, kq_scale, il);384 cb(attn, "attn_pre_o", il);385 386 cur = build_switched_lora_mm(layer.wo, sl.a_o, sl.b_o, attn, adapter_ids);387 cb(cur, "attn_out", il);388 return cur;389}390 391ggml_tensor * llama_model_granite_switch::graph::build_layer_ffn(392 ggml_tensor * cur,393 ggml_tensor * inpSA,394 ggml_tensor * adapter_ids,395 const llama_model & model,396 const int il) {397 398 const auto & layer = model.layers[il];399 const auto & sl = layer.switch_lora;400 401 if (hparams.f_residual_scale) {402 cur = ggml_scale(ctx0, cur, hparams.f_residual_scale);403 }404 ggml_tensor * ffn_inp = ggml_add(ctx0, cur, inpSA);405 cb(ffn_inp, "ffn_inp", il);406 407 cur = build_norm(ffn_inp, layer.ffn_norm, NULL, LLM_NORM_RMS, il);408 cb(cur, "ffn_norm", il);409 410 ggml_tensor * g = build_switched_lora_mm(layer.ffn_gate, sl.a_gate, sl.b_gate, cur, adapter_ids);411 ggml_tensor * u = build_switched_lora_mm(layer.ffn_up, sl.a_up, sl.b_up, cur, adapter_ids);412 g = ggml_silu(ctx0, g);413 ggml_tensor * gu = ggml_mul(ctx0, g, u);414 cur = build_switched_lora_mm(layer.ffn_down, sl.a_down, sl.b_down, gu, adapter_ids);415 cb(cur, "ffn_out", il);416 417 if (hparams.f_residual_scale) {418 cur = ggml_scale(ctx0, cur, hparams.f_residual_scale);419 }420 cur = ggml_add(ctx0, cur, ffn_inp);421 422 cur = build_cvec(cur, il);423 cb(cur, "l_out", il);424 425 return cur;426}427 