Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#include "llama-hparams.h"2 3#include "ggml.h"4 5#include <algorithm>6#include <cassert>7 8void llama_hparams::set_swa_pattern(uint32_t n_pattern, bool dense_first) {9 if (dense_first) {10 for (uint32_t il = 0; il < n_layer(); ++il) {11 is_swa_impl[il] = n_pattern == 0 || (il % n_pattern != 0);12 }13 } else {14 for (uint32_t il = 0; il < n_layer(); ++il) {15 is_swa_impl[il] = n_pattern == 0 || (il % n_pattern < (n_pattern - 1));16 }17 }18 19 for (uint32_t il = n_layer(); il < n_layer_all; ++il) {20 is_swa_impl[il] = false;21 }22}23 24void llama_hparams::set_recr_pattern(uint32_t n_pattern, bool dense_first) {25 if (dense_first) {26 for (uint32_t il = 0; il < n_layer(); ++il) {27 is_recr_impl[il] = n_pattern == 0 || (il % n_pattern != 0);28 }29 } else {30 for (uint32_t il = 0; il < n_layer(); ++il) {31 is_recr_impl[il] = n_pattern == 0 || (il % n_pattern < (n_pattern - 1));32 }33 }34 35 for (uint32_t il = n_layer(); il < n_layer_all; ++il) {36 is_recr_impl[il] = false;37 }38}39 40bool llama_hparams::is_swa_any() const {41 for (uint32_t il = 0; il < n_layer_all; ++il) {42 if (is_swa_impl[il]) {43 return true;44 }45 }46 47 return false;48}49 50uint32_t llama_hparams::n_head(uint32_t il) const {51 if (il < n_layer_all) {52 return n_head_arr[il];53 }54 55 GGML_ABORT("fatal error");56}57 58uint32_t llama_hparams::n_head_kv(uint32_t il) const {59 if (il < n_layer_all) {60 return n_head_kv_arr[il];61 }62 63 GGML_ABORT("fatal error");64}65 66uint32_t llama_hparams::n_ff(uint32_t il) const {67 if (il < n_layer_all) {68 return n_ff_arr[il];69 }70 71 GGML_ABORT("fatal error");72}73 74uint32_t llama_hparams::n_gqa(uint32_t il) const {75 const uint32_t n_head = this->n_head(il);76 const uint32_t n_head_kv = this->n_head_kv(il);77 78 if (n_head_kv == 0) {79 return 0;80 }81 82 return n_head/n_head_kv;83}84 85uint32_t llama_hparams::n_rot(uint32_t il) const {86 if (il < n_layer_all) {87 return is_swa(il) ? n_rot_swa : n_rot_full;88 }89 90 GGML_ABORT("fatal error");91}92 93uint32_t llama_hparams::n_embd_inp() const {94 if (n_embd_inp_impl > 0) {95 return n_embd_inp_impl;96 }97 98 uint32_t n_embd_inp = n_embd;99 100 if (n_deepstack_layers > 0) {101 n_embd_inp += n_embd * n_deepstack_layers;102 }103 104 return n_embd_inp;105}106 107uint32_t llama_hparams::n_embd_inp_enc() const {108 return n_embd_inp_enc_impl > 0 ? n_embd_inp_enc_impl : n_embd_inp();109}110 111uint32_t llama_hparams::n_embd_out() const {112 return n_embd_out_impl > 0 ? n_embd_out_impl : n_embd;113}114 115uint32_t llama_hparams::n_embd_head_k(uint32_t il) const {116 if (il < n_layer_all) {117 return is_swa(il) ? n_embd_head_k_swa : n_embd_head_k_full;118 }119 120 GGML_ABORT("fatal error");121}122 123uint32_t llama_hparams::n_embd_head_v(uint32_t il) const {124 if (il < n_layer_all) {125 return is_swa(il) ? n_embd_head_v_swa : n_embd_head_v_full;126 }127 128 GGML_ABORT("fatal error");129}130 131uint32_t llama_hparams::n_embd_k_gqa(uint32_t il) const {132 const uint32_t n_head_kv = this->n_head_kv(il);133 134 return n_embd_head_k(il) * n_head_kv;135}136 137uint32_t llama_hparams::n_embd_v_gqa(uint32_t il) const {138 const uint32_t n_head_kv = this->n_head_kv(il);139 140 return n_embd_head_v(il) * n_head_kv;141}142 143bool llama_hparams::is_n_embd_k_gqa_variable() const {144 const uint32_t val = n_embd_k_gqa();145 for (uint32_t il = 0; il < n_layer_all; ++il) {146 if (val != n_embd_k_gqa(il)) {147 return true;148 }149 }150 151 return false;152}153 154bool llama_hparams::is_n_embd_v_gqa_variable() const {155 const uint32_t val = n_embd_v_gqa();156 for (uint32_t il = 0; il < n_layer_all; ++il) {157 if (val != n_embd_v_gqa(il)) {158 return true;159 }160 }161 162 return false;163}164 165uint32_t llama_hparams::n_embd_k_gqa_max() const {166 uint32_t val = n_embd_k_gqa();167 for (uint32_t il = 0; il < n_layer_all; ++il) {168 val = std::max(val, n_embd_k_gqa(il));169 }170 171 return val;172}173 174uint32_t llama_hparams::n_embd_v_gqa_max() const {175 uint32_t val = n_embd_v_gqa();176 for (uint32_t il = 0; il < n_layer_all; ++il) {177 val = std::max(val, n_embd_v_gqa(il));178 }179 180 return val;181}182 183uint32_t llama_hparams::n_embd_r() const {184 if (wkv_head_size != 0) {185 // for RWKV models186 return token_shift_count * n_embd;187 }188 189 if (n_shortconv_l_cache != 0) {190 // for LFM2 models191 return n_embd * (n_shortconv_l_cache - 1);192 }193 194 if (n_embd_head_kda != 0) {195 // for Kimi KDA layers196 // Conv state for Q, K, V: 3 * (d_conv - 1) * n_head * head_dim197 const uint32_t d_inner = n_head() * n_embd_head_kda; // 32 * 128 = 4096198 return 3 * (ssm_d_conv > 0 ? ssm_d_conv - 1 : 3) * d_inner;199 }200 201 // TODO: maybe support other convolution strides than 1202 // NOTE: since the first column of the conv_state is shifted out each time, it's not actually needed203 // Corresponds to Mamba's conv_states size204 return (ssm_d_conv > 0 ? ssm_d_conv - 1 : 0) * (ssm_d_inner + 2*ssm_n_group*ssm_d_state);205}206 207uint32_t llama_hparams::n_embd_s() const {208 if (wkv_head_size != 0) {209 // corresponds to RWKV's wkv_states size210 return n_embd * wkv_head_size;211 }212 213 if (n_embd_head_kda != 0) {214 // for Kimi KDA layers215 // Full recurrent state: head_dim * head_dim * n_head216 // h tensor shape for delta attention: [head_dim, head_dim, n_head]217 return n_embd_head_kda * n_embd_head_kda * n_head(); // 128 * 128 * 32 = 524288218 }219 220 // corresponds to Mamba's ssm_states size221 return ssm_d_state * ssm_d_inner;222}223 224bool llama_hparams::is_recr(uint32_t il) const {225 if (il < n_layer_all) {226 return is_recr_impl[il];227 }228 229 GGML_ABORT("%s: il (%u) out of bounds (n_layer_all: %u)\n", __func__, il, n_layer_all);230}231 232uint32_t llama_hparams::n_pos_per_embd() const {233 return rope_type == LLAMA_ROPE_TYPE_MROPE || rope_type == LLAMA_ROPE_TYPE_IMROPE ? 4 : 1;234}235 236bool llama_hparams::is_swa(uint32_t il) const {237 if (il < n_layer_all) {238 return is_swa_impl[il];239 }240 241 GGML_ABORT("%s: il (%u) out of bounds (n_layer_all: %u)\n", __func__, il, n_layer_all);242}243 244bool llama_hparams::is_mla() const {245 assert((n_embd_head_k_mla_impl == 0 && n_embd_head_v_mla_impl == 0) ||246 (n_embd_head_k_mla_impl != 0 && n_embd_head_v_mla_impl != 0));247 248 return n_embd_head_k_mla_impl != 0 && n_embd_head_v_mla_impl != 0;249}250 251bool llama_hparams::is_indexer_full(uint32_t il) const {252 if (il < n_layer()) {253 return is_indexer_full_impl[il];254 }255 256 GGML_ABORT("%s: il (%u) out of bounds (n_layer: %u)\n", __func__, il, n_layer());257}258 259uint32_t llama_hparams::n_embd_head_k_mla() const {260 return is_mla() ? n_embd_head_k_mla_impl : n_embd_head_k();261}262 263uint32_t llama_hparams::n_embd_head_v_mla() const {264 return is_mla() ? n_embd_head_v_mla_impl : n_embd_head_v();265}266 267bool llama_hparams::has_kv(uint32_t il) const {268 if (n_layer_kv_from_start >= 0) {269 if (il < (uint32_t) n_layer_kv_from_start) {270 return true;271 }272 273 return false;274 }275 276 // by default, all layers have kv277 return true;278}279 280bool llama_hparams::has_rope(uint32_t il) const {281 // the router layer stores adapter routing signal, not positional info,282 // so it must not be RoPE-shifted283 if (router_layer >= 0 && (int32_t) il == router_layer) {284 return false;285 }286 287 return true;288}289 290uint32_t llama_hparams::n_layer() const {291 return n_layer_all - n_layer_nextn;292}293 294bool llama_hparams::use_mrope() const {295 return rope_sections[0] > 0 && rope_sections[1] > 0;296}297 