Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#include "models.h"2 3static constexpr int PARAKEET_LOCAL_ATTN_THRESHOLD = 8192;4static constexpr int PARAKEET_LOCAL_ATTN_WINDOW = 128;5 6// conv subsampling + conformer encoder7ggml_cgraph * clip_graph_parakeet::build() {8 9 // Conv subsampling10 ggml_tensor * inp = build_inp_raw(1);11 inp = ggml_cont(ctx0, ggml_transpose(ctx0, inp));12 13 // [freq, time, channels, batch]14 ggml_tensor * cur = ggml_conv_2d(ctx0, model.pre_encode_conv_X_w[0], inp, 2, 2, 1, 1, 1, 1);15 cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[0]);16 cb(cur, "pre_conv_0", -1);17 18 cur = ggml_relu(ctx0, cur);19 cb(cur, "pre_conv_0_relu", -1);20 21 // [freq, time, channels, batch]22 cur = ggml_conv_2d_dw_direct(ctx0, model.pre_encode_conv_X_w[2], cur, 2, 2, 1, 1, 1, 1);23 cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[2]);24 cb(cur, "pre_conv_2", -1);25 26 // [freq, time, channels, batch]27 cur = ggml_conv_2d(ctx0, model.pre_encode_conv_X_w[3], cur, 1, 1, 0, 0, 1, 1);28 cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[3]);29 cb(cur, "pre_conv_3", -1);30 31 cur = ggml_relu(ctx0, cur);32 cb(cur, "pre_conv_3_relu", -1);33 34 // [freq, time, channels, batch]35 cur = ggml_conv_2d_dw_direct(ctx0, model.pre_encode_conv_X_w[5], cur, 2, 2, 1, 1, 1, 1);36 cb(cur, "pre_conv_5_direct", -1);37 cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[5]);38 cb(cur, "pre_conv_5", -1);39 40 // [freq, time, channels, batch]41 cur = ggml_conv_2d(ctx0, model.pre_encode_conv_X_w[6], cur, 1, 1, 0, 0, 1, 1);42 cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[6]);43 cb(cur, "pre_conv_6", -1);44 45 cur = ggml_relu(ctx0, cur);46 cb(cur, "pre_conv_6_relu", -1);47 48 // [freq, time, chan]49 cur = ggml_permute(ctx0, cur, 0, 2, 1, 3);50 // [freq, chan, time]51 cur = ggml_cont(ctx0, cur);52 53 const int n_freq = cur->ne[0];54 const int n_chan = cur->ne[1];55 const int n_frames = cur->ne[2];56 57 // [freq, time, chan, batch] -> [(freq * chan), time]58 cur = ggml_reshape_2d(ctx0, cur, n_freq * n_chan, n_frames);59 60 cur = build_mm(model.pre_encode_out_w, cur);61 cur = ggml_add(ctx0, cur, model.pre_encode_out_b);62 63 ggml_set_name(cur, "pre_enc_out");64 65 // Encoder66 67 const auto & hparams = model.hparams;68 const int n_layer = hparams.n_layer;69 const int n_state = hparams.n_embd;70 const float fc_factor = 0.5f;71 72 const int n_time = cur->ne[1];73 const bool local_attn = n_time > PARAKEET_LOCAL_ATTN_THRESHOLD;74 const int att_left = local_attn ? PARAKEET_LOCAL_ATTN_WINDOW : n_time - 1;75 const int att_right = local_attn ? PARAKEET_LOCAL_ATTN_WINDOW : n_time - 1;76 const int window_size = local_attn ? att_left + att_right + 1 : 2 * n_time - 1;77 const int d_half = n_state / 2;78 const int mask_dim = local_attn ? window_size : n_time;79 80 // mask [key, n_time]81 struct ggml_tensor * attn_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, mask_dim, n_time);82 ggml_set_name(attn_mask, "attn_mask");83 ggml_set_input(attn_mask);84 85 struct ggml_tensor * local_mask = nullptr;86 if (local_attn) {87 const int chunk = att_left + att_right;88 local_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, chunk + window_size - 1, chunk);89 ggml_set_name(local_mask, "local_mask");90 ggml_set_input(local_mask);91 }92 93 struct ggml_tensor * pos_freqs = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, d_half);94 ggml_set_name(pos_freqs, "pos_freqs");95 ggml_set_input(pos_freqs);96 97 struct ggml_tensor * rel_positions = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, 1, window_size);98 ggml_set_name(rel_positions, "rel_positions");99 ggml_set_input(rel_positions);100 101 struct ggml_tensor * freqs = ggml_repeat_4d(ctx0, pos_freqs, d_half, window_size, 1, 1);102 struct ggml_tensor * theta = ggml_mul(ctx0, freqs, rel_positions);103 104 struct ggml_tensor * sin = ggml_reshape_3d(ctx0, ggml_sin(ctx0, theta), 1, d_half, window_size);105 struct ggml_tensor * cos = ggml_reshape_3d(ctx0, ggml_cos(ctx0, theta), 1, d_half, window_size);106 struct ggml_tensor * pos_emb = ggml_reshape_2d(ctx0, ggml_cont(ctx0, ggml_concat(ctx0, sin, cos, 0)), n_state, window_size);107 ggml_set_name(pos_emb, "pos_emb");108 109 for (int il = 0; il < n_layer; ++il) {110 const auto & layer = model.layers[il];111 // FFN1112 {113 struct ggml_tensor * residual = cur;114 ggml_format_name(cur, "enc_%d_res", il);115 116 // norm117 cur = ggml_norm(ctx0, cur, hparams.eps);118 cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.ff_norm_w), layer.ff_norm_b);119 ggml_format_name(cur, "enc_%d_ffn_norm_1", il);120 121 cur = build_ffn(cur, layer.ff_up_w, nullptr, nullptr, nullptr, layer.ff_down_w, nullptr, FFN_SILU, il);122 ggml_format_name(cur, "enc_%d_ffn_1", il);123 124 cur = ggml_add(ctx0, residual, ggml_scale(ctx0, cur, fc_factor));125 ggml_format_name(cur, "enc_%d_res_ffn", il);126 }127 128 // self attention block using relative positional encoding from model.position_embedding.129 {130 // [feat, time_frames, 1, 1]131 struct ggml_tensor * residual = cur;132 133 cur = ggml_norm(ctx0, cur, hparams.eps);134 cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.ln_1_w), layer.ln_1_b);135 ggml_format_name(cur, "enc_%d_attn_norm", il);136 137 const int n_head = hparams.n_head;138 const int d_head = n_state / n_head;139 140 // [feat, time_frames, 1, 1]141 struct ggml_tensor * Q_cur = build_mm(layer.q_w, cur);142 struct ggml_tensor * K_cur = build_mm(layer.k_w, cur);143 struct ggml_tensor * V_cur = build_mm(layer.v_w, cur);144 145 // [d_head, n_heads, n_time, 1]146 Q_cur = ggml_reshape_3d(ctx0, Q_cur, d_head, n_head, n_time);147 K_cur = ggml_reshape_3d(ctx0, K_cur, d_head, n_head, n_time);148 V_cur = ggml_reshape_3d(ctx0, V_cur, d_head, n_head, n_time);149 150 // [n_state, window_size]151 struct ggml_tensor * pos = build_mm(layer.linear_pos_w, pos_emb);152 // [feat, head, window_size, 1]153 pos = ggml_reshape_3d(ctx0, pos, d_head, n_head, pos_emb->ne[1]);154 // [feat, window_size, head, 1]155 pos = ggml_cont(ctx0, ggml_permute(ctx0, pos, 0, 2, 1, 3));156 ggml_format_name(pos, "enc_%d_attn_pos", il);157 158 if (local_attn) {159 const int chunk = att_left + att_right;160 const int n_group = (n_time + chunk - 1) / chunk;161 const int n_time_padded = n_group * chunk;162 const int n_kv_chunk = chunk + window_size - 1;163 const int n_kv_dense = n_kv_chunk * n_group;164 const bool need_padding = n_time_padded > n_time;165 166 Q_cur = ggml_cont(ctx0, ggml_permute(ctx0, Q_cur, 0, 2, 1, 3));167 K_cur = ggml_cont(ctx0, ggml_permute(ctx0, K_cur, 0, 2, 1, 3));168 V_cur = ggml_cont(ctx0, ggml_permute(ctx0, V_cur, 0, 2, 1, 3));169 170 // content bias171 struct ggml_tensor * bias_u = ggml_reshape_3d(ctx0, layer.pos_bias_u, d_head, 1, n_head);172 struct ggml_tensor * Q_u = ggml_add(ctx0, Q_cur, bias_u);173 174 // position bias175 struct ggml_tensor * bias_v = ggml_reshape_3d(ctx0, layer.pos_bias_v, d_head, 1, n_head);176 struct ggml_tensor * Q_v = ggml_add(ctx0, Q_cur, bias_v);177 178 // right pad the time dimension179 struct ggml_tensor * Q_u_padded = need_padding ?180 ggml_pad_ext(ctx0, Q_u, 0, 0, 0, n_time_padded - n_time, 0, 0, 0, 0) : Q_u;181 Q_u_padded = ggml_reshape_4d(ctx0, Q_u_padded, d_head, chunk, n_group, n_head);182 183 // pad front and back for the first and last time frames184 struct ggml_tensor * K_padded = ggml_pad_ext(ctx0, K_cur, 0, 0, att_left, att_right, 0, 0, 0, 0);185 if (n_kv_dense > K_padded->ne[1]) {186 K_padded = ggml_pad_ext(ctx0, K_padded, 0, 0, 0, n_kv_dense - K_padded->ne[1], 0, 0, 0, 0);187 }188 189 // sliding window view: each group spans n_kv_chunk keys but steps by chunk190 struct ggml_tensor * K_chunk = ggml_view_4d(ctx0, K_padded,191 d_head, n_kv_chunk, n_group, n_head,192 K_padded->nb[1],193 (size_t) chunk * K_padded->nb[1],194 K_padded->nb[2],195 0);196 K_chunk = ggml_cont(ctx0, K_chunk);197 198 struct ggml_tensor * content_scores = ggml_mul_mat(ctx0, K_chunk, Q_u_padded);199 200 // trim the dense output down to window_size scores per query201 content_scores = ggml_view_4d(ctx0, content_scores,202 window_size, chunk, n_group, n_head,203 (size_t) (chunk + window_size) * content_scores->nb[0],204 content_scores->nb[2],205 content_scores->nb[3],206 0);207 content_scores = ggml_cont(ctx0, content_scores);208 209 // ungroup: [window_size, n_time_padded, n_head]210 content_scores = ggml_reshape_3d(ctx0, content_scores, window_size, n_time_padded, n_head);211 if (need_padding) {212 content_scores = ggml_view_3d(ctx0, content_scores,213 window_size, n_time, n_head,214 content_scores->nb[1],215 content_scores->nb[2],216 0);217 }218 219 // Q_v: [d_head, time, head]220 Q_v = ggml_cont(ctx0, ggml_permute(ctx0, Q_v, 0, 2, 1, 3));221 struct ggml_tensor * rel_pos_scores = ggml_mul_mat(ctx0, pos, Q_v);222 223 struct ggml_tensor * attn_scores = ggml_add(ctx0, content_scores, rel_pos_scores);224 attn_scores = ggml_soft_max_ext(ctx0, attn_scores, attn_mask, 1.0f / std::sqrt(d_head), 0.0f);225 ggml_format_name(attn_scores, "enc_%d_attn_probs", il);226 227 // expand probs back to n_kv_chunk width for the V matmul228 struct ggml_tensor * probs_padded = need_padding ?229 ggml_pad_ext(ctx0, attn_scores, 0, 0, 0, n_time_padded - n_time, 0, 0, 0, 0) : attn_scores;230 231 probs_padded = ggml_reshape_4d(ctx0, probs_padded, window_size, chunk, n_group, n_head);232 probs_padded = ggml_pad_ext(ctx0, probs_padded, 0, chunk, 0, 0, 0, 0, 0, 0);233 probs_padded = ggml_view_4d(ctx0, probs_padded,234 n_kv_chunk, chunk, n_group, n_head,235 (size_t) n_kv_chunk * probs_padded->nb[0],236 probs_padded->nb[2],237 probs_padded->nb[3],238 0);239 probs_padded = ggml_cont(ctx0, probs_padded);240 probs_padded = ggml_mul(ctx0, probs_padded, local_mask);241 242 struct ggml_tensor * V_padded = ggml_pad_ext(ctx0, V_cur, 0, 0, att_left, att_right, 0, 0, 0, 0);243 if (n_kv_dense > V_padded->ne[1]) {244 V_padded = ggml_pad_ext(ctx0, V_padded, 0, 0, 0, n_kv_dense - V_padded->ne[1], 0, 0, 0, 0);245 }246 V_padded = ggml_cont(ctx0, ggml_transpose(ctx0, V_padded));247 248 struct ggml_tensor * V_chunk = ggml_view_4d(ctx0, V_padded,249 n_kv_chunk, d_head, n_group, n_head,250 V_padded->nb[1],251 (size_t) chunk * V_padded->nb[0],252 V_padded->nb[2],253 0);254 V_chunk = ggml_cont(ctx0, V_chunk);255 256 cur = ggml_mul_mat(ctx0, V_chunk, probs_padded);257 cur = ggml_reshape_3d(ctx0, cur, d_head, n_time_padded, n_head);258 if (need_padding) {259 cur = ggml_view_3d(ctx0, cur, d_head, n_time, n_head, cur->nb[1], cur->nb[2], 0);260 }261 cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 0, 2, 1, 3));262 cur = ggml_reshape_2d(ctx0, cur, n_state, n_time);263 cur = build_mm(layer.o_w, cur);264 } else {265 // full attention266 struct ggml_tensor * Q_u = ggml_add(ctx0, Q_cur, layer.pos_bias_u);267 ggml_format_name(Q_u, "enc_%d_attn_q_u", il);268 269 struct ggml_tensor * K_prep = ggml_permute(ctx0, K_cur, 0, 2, 1, 3);270 struct ggml_tensor * Q_prep = ggml_permute(ctx0, Q_u, 0, 2, 1, 3);271 struct ggml_tensor * content_scores = ggml_mul_mat(ctx0, K_prep, Q_prep);272 ggml_format_name(content_scores, "enc_%d_attn_content_scores", il);273 274 struct ggml_tensor * Q_v = ggml_add(ctx0, Q_cur, layer.pos_bias_v);275 ggml_format_name(Q_v, "enc_%d_attn_q_v", il);276 277 Q_v = ggml_permute(ctx0, Q_v, 0, 2, 1, 3);278 Q_v = ggml_cont(ctx0, Q_v);279 ggml_format_name(Q_v, "enc_%d_attn_q_v_perm", il);280 281 struct ggml_tensor * rel_pos_scores = ggml_mul_mat(ctx0, pos, Q_v);282 ggml_format_name(rel_pos_scores, "enc_%d_attn_rel_pos", il);283 284 // Relative positional shift285 {286 const auto pos_window = rel_pos_scores->ne[0];287 const auto n_frame = rel_pos_scores->ne[1];288 const auto n_head = rel_pos_scores->ne[2];289 290 rel_pos_scores = ggml_pad(ctx0, rel_pos_scores, 1, 0, 0, 0);291 rel_pos_scores = ggml_roll(ctx0, rel_pos_scores, 1, 0, 0, 0);292 293 rel_pos_scores = ggml_reshape_3d(ctx0, rel_pos_scores, n_frame, pos_window + 1, n_head);294 rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);295 ggml_format_name(rel_pos_scores, "enc_%d_attn_rel_pos_reshaped", il);296 297 int center = pos_window / 2;298 size_t offset = rel_pos_scores->nb[0] * (center+1);299 300 rel_pos_scores = ggml_view_3d(ctx0, rel_pos_scores,301 n_frame, pos_window, n_head,302 (pos_window) * 4,303 rel_pos_scores->nb[2],304 offset);305 rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);306 ggml_format_name(rel_pos_scores, "enc_%d_attn_rel_pos_shifted", il);307 308 rel_pos_scores = ggml_view_3d(ctx0, rel_pos_scores,309 content_scores->ne[0],310 content_scores->ne[1],311 rel_pos_scores->ne[2],312 rel_pos_scores->nb[1],313 rel_pos_scores->nb[2],314 0);315 rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);316 ggml_format_name(rel_pos_scores, "enc_%d_attn_rel_pos_shifted_view", il);317 }318 319 struct ggml_tensor * attn_scores = ggml_add(ctx0, content_scores, rel_pos_scores);320 ggml_format_name(attn_scores, "enc_%d_attn_scores", il);321 attn_scores = ggml_scale(ctx0, attn_scores, 1.0f / std::sqrt(d_head));322 attn_scores = ggml_add(ctx0, attn_scores, attn_mask);323 ggml_format_name(attn_scores, "enc_%d_attn_scores_scaled", il);324 325 struct ggml_tensor * probs = ggml_soft_max(ctx0, attn_scores);326 ggml_format_name(probs, "enc_%d_attn_probs", il);327 328 V_cur = ggml_cont(ctx0, ggml_permute(ctx0, V_cur, 1, 2, 0, 3));329 ggml_format_name(V_cur, "enc_%d_attn_v_cur", il);330 cur = ggml_mul_mat(ctx0, probs, V_cur);331 ggml_format_name(cur, "enc_%d_attn_inp", il);332 333 cur = ggml_permute(ctx0, cur, 2, 0, 1, 3);334 cur = ggml_cont_2d(ctx0, cur, n_state, n_time);335 cur = build_mm(layer.o_w, cur);336 }337 ggml_format_name(cur, "enc_%d_attn_out", il);338 339 cur = ggml_add(ctx0, residual, cur);340 ggml_format_name(cur, "enc_%d_attn_res", il);341 }342 343 // Convolution344 {345 struct ggml_tensor * residual = cur;346 ggml_format_name(cur, "enc_%d_residual_conv", il);347 348 cur = ggml_norm(ctx0, cur, hparams.eps);349 cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.norm_conv_w), layer.norm_conv_b);350 ggml_format_name(cur, "enc_%d_norm_conv", il);351 352 // pointwise 1d convolution:353 cur = build_mm(layer.conv_pw1_w, cur);354 ggml_format_name(cur, "enc_%d_conv_pw1", il);355 356 {357 int64_t d = cur->ne[0] / 2;358 struct ggml_tensor * signal = ggml_view_2d(ctx0, cur, d, cur->ne[1], cur->nb[1], 0);359 struct ggml_tensor * gate = ggml_view_2d(ctx0, cur, d, cur->ne[1], cur->nb[1], d * cur->nb[0]);360 361 cur = ggml_mul(ctx0, signal, ggml_sigmoid(ctx0, gate));362 ggml_format_name(cur, "enc_%d_conv_glu", il);363 }364 365 cur = ggml_cont(ctx0, ggml_transpose(ctx0, cur));366 367 // use ggml_ssm_conv for f32 precision368 const int dw_pad = (hparams.audio_conv_kernel_size - 1) / 2;369 cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);370 cur = ggml_roll(ctx0, cur, dw_pad, 0, 0, 0);371 cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);372 ggml_format_name(cur, "enc_%d_conv_dw_pad", il);373 374 cur = ggml_ssm_conv(ctx0, cur, layer.conv_dw_w);375 ggml_format_name(cur, "enc_%d_conv_1d_dw", il);376 377 cur = ggml_sub(ctx0, cur, layer.conv_norm_mean);378 struct ggml_tensor * std = ggml_sqrt(ctx0, layer.conv_norm_var);379 cur = ggml_div(ctx0, cur, std);380 cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.conv_norm_w), layer.conv_norm_b);381 ggml_format_name(cur, "enc_%d_conv_bn", il);382 383 cur = ggml_silu(ctx0, cur);384 ggml_format_name(cur, "enc_%d_conv_silu", il);385 386 cur = build_mm(layer.conv_pw2_w, cur);387 ggml_format_name(cur, "enc_%d_conv_pw2", il);388 389 cur = ggml_add(ctx0, residual, cur);390 ggml_format_name(cur, "enc_%d_conv_res", il);391 }392 393 // FFN2394 {395 struct ggml_tensor * residual = cur;396 cur = ggml_norm(ctx0, cur, hparams.eps);397 cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.ff_norm_1_w), layer.ff_norm_1_b);398 ggml_format_name(cur, "enc_%d_ffn_norm_2", il);399 400 cur = build_ffn(cur, layer.ff_up_1_w, nullptr, nullptr, nullptr, layer.ff_down_1_w, nullptr, FFN_SILU, il);401 cur = ggml_add(ctx0, residual, ggml_scale(ctx0, cur, 0.5));402 ggml_format_name(cur, "enc_%d_ffn_res", il);403 }404 405 cur = ggml_norm(ctx0, cur, hparams.eps);406 cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.ln_2_w), layer.ln_2_b);407 }408 409 cb(cur, "encoder_out", -1);410 411 cur = ggml_rms_norm(ctx0, cur, 1e-6);412 cur = ggml_mul(ctx0, cur, model.mm_norm_pre_w);413 cb(cur, "sound_projection.norm", -1);414 415 cur = build_ffn(cur, model.mm_0_w, model.mm_0_b, nullptr, nullptr, model.mm_1_w, model.mm_1_b, FFN_RELU_SQR, -1);416 cb(cur, "projected", -1);417 418 ggml_build_forward_expand(gf, cur);419 420 return gf;421}422 