Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
parakeet.cpp422 linesDownload Raw Back to models
1#include "models.h"2 3static constexpr int PARAKEET_LOCAL_ATTN_THRESHOLD = 8192;4static constexpr int PARAKEET_LOCAL_ATTN_WINDOW    = 128;5 6// conv subsampling + conformer encoder7ggml_cgraph * clip_graph_parakeet::build() {8 9    // Conv subsampling10    ggml_tensor * inp = build_inp_raw(1);11    inp = ggml_cont(ctx0, ggml_transpose(ctx0, inp));12 13    // [freq, time, channels, batch]14    ggml_tensor * cur = ggml_conv_2d(ctx0, model.pre_encode_conv_X_w[0], inp, 2, 2, 1, 1, 1, 1);15    cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[0]);16    cb(cur, "pre_conv_0", -1);17 18    cur = ggml_relu(ctx0, cur);19    cb(cur, "pre_conv_0_relu", -1);20 21    // [freq, time, channels, batch]22    cur = ggml_conv_2d_dw_direct(ctx0, model.pre_encode_conv_X_w[2], cur, 2, 2, 1, 1, 1, 1);23    cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[2]);24    cb(cur, "pre_conv_2", -1);25 26    // [freq, time, channels, batch]27    cur = ggml_conv_2d(ctx0, model.pre_encode_conv_X_w[3], cur, 1, 1, 0, 0, 1, 1);28    cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[3]);29    cb(cur, "pre_conv_3", -1);30 31    cur = ggml_relu(ctx0, cur);32    cb(cur, "pre_conv_3_relu", -1);33 34    // [freq, time, channels, batch]35    cur = ggml_conv_2d_dw_direct(ctx0, model.pre_encode_conv_X_w[5], cur, 2, 2, 1, 1, 1, 1);36    cb(cur, "pre_conv_5_direct", -1);37    cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[5]);38    cb(cur, "pre_conv_5", -1);39 40    // [freq, time, channels, batch]41    cur = ggml_conv_2d(ctx0, model.pre_encode_conv_X_w[6], cur, 1, 1, 0, 0, 1, 1);42    cur = ggml_add(ctx0, cur, model.pre_encode_conv_X_b[6]);43    cb(cur, "pre_conv_6", -1);44 45    cur = ggml_relu(ctx0, cur);46    cb(cur, "pre_conv_6_relu", -1);47 48    // [freq, time, chan]49    cur = ggml_permute(ctx0, cur, 0, 2, 1, 3);50    // [freq, chan, time]51    cur = ggml_cont(ctx0, cur);52 53    const int n_freq   = cur->ne[0];54    const int n_chan   = cur->ne[1];55    const int n_frames = cur->ne[2];56 57    // [freq, time, chan, batch] -> [(freq * chan), time]58    cur = ggml_reshape_2d(ctx0, cur, n_freq * n_chan, n_frames);59 60    cur = build_mm(model.pre_encode_out_w, cur);61    cur = ggml_add(ctx0, cur, model.pre_encode_out_b);62 63    ggml_set_name(cur, "pre_enc_out");64 65    // Encoder66 67    const auto & hparams  = model.hparams;68    const int n_layer     = hparams.n_layer;69    const int n_state     = hparams.n_embd;70    const float fc_factor = 0.5f;71 72    const int  n_time      = cur->ne[1];73    const bool local_attn  = n_time > PARAKEET_LOCAL_ATTN_THRESHOLD;74    const int  att_left    = local_attn ? PARAKEET_LOCAL_ATTN_WINDOW : n_time - 1;75    const int  att_right   = local_attn ? PARAKEET_LOCAL_ATTN_WINDOW : n_time - 1;76    const int  window_size = local_attn ? att_left + att_right + 1 : 2 * n_time - 1;77    const int  d_half      = n_state / 2;78    const int  mask_dim    = local_attn ? window_size : n_time;79 80    // mask [key, n_time]81    struct ggml_tensor * attn_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, mask_dim, n_time);82    ggml_set_name(attn_mask, "attn_mask");83    ggml_set_input(attn_mask);84 85    struct ggml_tensor * local_mask = nullptr;86    if (local_attn) {87        const int chunk = att_left + att_right;88        local_mask = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, chunk + window_size - 1, chunk);89        ggml_set_name(local_mask, "local_mask");90        ggml_set_input(local_mask);91    }92 93    struct ggml_tensor * pos_freqs = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, d_half);94    ggml_set_name(pos_freqs, "pos_freqs");95    ggml_set_input(pos_freqs);96 97    struct ggml_tensor * rel_positions = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, 1, window_size);98    ggml_set_name(rel_positions, "rel_positions");99    ggml_set_input(rel_positions);100 101    struct ggml_tensor * freqs = ggml_repeat_4d(ctx0, pos_freqs, d_half, window_size, 1, 1);102    struct ggml_tensor * theta = ggml_mul(ctx0, freqs, rel_positions);103 104    struct ggml_tensor * sin = ggml_reshape_3d(ctx0, ggml_sin(ctx0, theta), 1, d_half, window_size);105    struct ggml_tensor * cos = ggml_reshape_3d(ctx0, ggml_cos(ctx0, theta), 1, d_half, window_size);106    struct ggml_tensor * pos_emb = ggml_reshape_2d(ctx0, ggml_cont(ctx0, ggml_concat(ctx0, sin, cos, 0)), n_state, window_size);107    ggml_set_name(pos_emb, "pos_emb");108 109    for (int il = 0; il < n_layer; ++il) {110        const auto & layer = model.layers[il];111        // FFN1112        {113            struct ggml_tensor * residual = cur;114            ggml_format_name(cur, "enc_%d_res", il);115 116            // norm117            cur = ggml_norm(ctx0, cur, hparams.eps);118            cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.ff_norm_w), layer.ff_norm_b);119            ggml_format_name(cur, "enc_%d_ffn_norm_1", il);120 121            cur = build_ffn(cur, layer.ff_up_w, nullptr, nullptr, nullptr, layer.ff_down_w, nullptr, FFN_SILU, il);122            ggml_format_name(cur, "enc_%d_ffn_1", il);123 124            cur = ggml_add(ctx0, residual, ggml_scale(ctx0, cur, fc_factor));125            ggml_format_name(cur, "enc_%d_res_ffn", il);126        }127 128        // self attention block using relative positional encoding from model.position_embedding.129        {130            // [feat, time_frames, 1, 1]131            struct ggml_tensor * residual = cur;132 133            cur = ggml_norm(ctx0, cur, hparams.eps);134            cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.ln_1_w), layer.ln_1_b);135            ggml_format_name(cur, "enc_%d_attn_norm", il);136 137            const int n_head = hparams.n_head;138            const int d_head = n_state / n_head;139 140            // [feat, time_frames, 1, 1]141            struct ggml_tensor * Q_cur = build_mm(layer.q_w, cur);142            struct ggml_tensor * K_cur = build_mm(layer.k_w, cur);143            struct ggml_tensor * V_cur = build_mm(layer.v_w, cur);144 145            // [d_head, n_heads, n_time, 1]146            Q_cur = ggml_reshape_3d(ctx0, Q_cur, d_head, n_head, n_time);147            K_cur = ggml_reshape_3d(ctx0, K_cur, d_head, n_head, n_time);148            V_cur = ggml_reshape_3d(ctx0, V_cur, d_head, n_head, n_time);149 150            // [n_state, window_size]151            struct ggml_tensor * pos = build_mm(layer.linear_pos_w, pos_emb);152            // [feat, head, window_size, 1]153            pos = ggml_reshape_3d(ctx0, pos, d_head, n_head, pos_emb->ne[1]);154            // [feat, window_size, head, 1]155            pos = ggml_cont(ctx0, ggml_permute(ctx0, pos, 0, 2, 1, 3));156            ggml_format_name(pos, "enc_%d_attn_pos", il);157 158            if (local_attn) {159                const int  chunk         = att_left + att_right;160                const int  n_group       = (n_time + chunk - 1) / chunk;161                const int  n_time_padded = n_group * chunk;162                const int  n_kv_chunk    = chunk + window_size - 1;163                const int  n_kv_dense    = n_kv_chunk * n_group;164                const bool need_padding  = n_time_padded > n_time;165 166                Q_cur = ggml_cont(ctx0, ggml_permute(ctx0, Q_cur, 0, 2, 1, 3));167                K_cur = ggml_cont(ctx0, ggml_permute(ctx0, K_cur, 0, 2, 1, 3));168                V_cur = ggml_cont(ctx0, ggml_permute(ctx0, V_cur, 0, 2, 1, 3));169 170                // content bias171                struct ggml_tensor * bias_u = ggml_reshape_3d(ctx0, layer.pos_bias_u, d_head, 1, n_head);172                struct ggml_tensor * Q_u = ggml_add(ctx0, Q_cur, bias_u);173 174                // position bias175                struct ggml_tensor * bias_v = ggml_reshape_3d(ctx0, layer.pos_bias_v, d_head, 1, n_head);176                struct ggml_tensor * Q_v = ggml_add(ctx0, Q_cur, bias_v);177 178                // right pad the time dimension179                struct ggml_tensor * Q_u_padded = need_padding ?180                    ggml_pad_ext(ctx0, Q_u, 0, 0, 0, n_time_padded - n_time, 0, 0, 0, 0) : Q_u;181                Q_u_padded = ggml_reshape_4d(ctx0, Q_u_padded, d_head, chunk, n_group, n_head);182 183                // pad front and back for the first and last time frames184                struct ggml_tensor * K_padded = ggml_pad_ext(ctx0, K_cur, 0, 0, att_left, att_right, 0, 0, 0, 0);185                if (n_kv_dense > K_padded->ne[1]) {186                    K_padded = ggml_pad_ext(ctx0, K_padded, 0, 0, 0, n_kv_dense - K_padded->ne[1], 0, 0, 0, 0);187                }188 189                // sliding window view: each group spans n_kv_chunk keys but steps by chunk190                struct ggml_tensor * K_chunk = ggml_view_4d(ctx0, K_padded,191                        d_head, n_kv_chunk, n_group, n_head,192                        K_padded->nb[1],193                        (size_t) chunk * K_padded->nb[1],194                        K_padded->nb[2],195                        0);196                K_chunk = ggml_cont(ctx0, K_chunk);197 198                struct ggml_tensor * content_scores = ggml_mul_mat(ctx0, K_chunk, Q_u_padded);199 200                // trim the dense output down to window_size scores per query201                content_scores = ggml_view_4d(ctx0, content_scores,202                        window_size, chunk, n_group, n_head,203                        (size_t) (chunk + window_size) * content_scores->nb[0],204                        content_scores->nb[2],205                        content_scores->nb[3],206                        0);207                content_scores = ggml_cont(ctx0, content_scores);208 209                // ungroup: [window_size, n_time_padded, n_head]210                content_scores = ggml_reshape_3d(ctx0, content_scores, window_size, n_time_padded, n_head);211                if (need_padding) {212                    content_scores = ggml_view_3d(ctx0, content_scores,213                            window_size, n_time, n_head,214                            content_scores->nb[1],215                            content_scores->nb[2],216                            0);217                }218 219                // Q_v: [d_head, time, head]220                Q_v = ggml_cont(ctx0, ggml_permute(ctx0, Q_v, 0, 2, 1, 3));221                struct ggml_tensor * rel_pos_scores = ggml_mul_mat(ctx0, pos, Q_v);222 223                struct ggml_tensor * attn_scores = ggml_add(ctx0, content_scores, rel_pos_scores);224                attn_scores = ggml_soft_max_ext(ctx0, attn_scores, attn_mask, 1.0f / std::sqrt(d_head), 0.0f);225                ggml_format_name(attn_scores, "enc_%d_attn_probs", il);226 227                // expand probs back to n_kv_chunk width for the V matmul228                struct ggml_tensor * probs_padded = need_padding ?229                    ggml_pad_ext(ctx0, attn_scores, 0, 0, 0, n_time_padded - n_time, 0, 0, 0, 0) : attn_scores;230 231                probs_padded = ggml_reshape_4d(ctx0, probs_padded, window_size, chunk, n_group, n_head);232                probs_padded = ggml_pad_ext(ctx0, probs_padded, 0, chunk, 0, 0, 0, 0, 0, 0);233                probs_padded = ggml_view_4d(ctx0, probs_padded,234                        n_kv_chunk, chunk, n_group, n_head,235                        (size_t) n_kv_chunk * probs_padded->nb[0],236                        probs_padded->nb[2],237                        probs_padded->nb[3],238                        0);239                probs_padded = ggml_cont(ctx0, probs_padded);240                probs_padded = ggml_mul(ctx0, probs_padded, local_mask);241 242                struct ggml_tensor * V_padded = ggml_pad_ext(ctx0, V_cur, 0, 0, att_left, att_right, 0, 0, 0, 0);243                if (n_kv_dense > V_padded->ne[1]) {244                    V_padded = ggml_pad_ext(ctx0, V_padded, 0, 0, 0, n_kv_dense - V_padded->ne[1], 0, 0, 0, 0);245                }246                V_padded = ggml_cont(ctx0, ggml_transpose(ctx0, V_padded));247 248                struct ggml_tensor * V_chunk = ggml_view_4d(ctx0, V_padded,249                        n_kv_chunk, d_head, n_group, n_head,250                        V_padded->nb[1],251                        (size_t) chunk * V_padded->nb[0],252                        V_padded->nb[2],253                        0);254                V_chunk = ggml_cont(ctx0, V_chunk);255 256                cur = ggml_mul_mat(ctx0, V_chunk, probs_padded);257                cur = ggml_reshape_3d(ctx0, cur, d_head, n_time_padded, n_head);258                if (need_padding) {259                    cur = ggml_view_3d(ctx0, cur, d_head, n_time, n_head, cur->nb[1], cur->nb[2], 0);260                }261                cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 0, 2, 1, 3));262                cur = ggml_reshape_2d(ctx0, cur, n_state, n_time);263                cur = build_mm(layer.o_w, cur);264            } else {265                // full attention266                struct ggml_tensor * Q_u = ggml_add(ctx0, Q_cur, layer.pos_bias_u);267                ggml_format_name(Q_u, "enc_%d_attn_q_u", il);268 269                struct ggml_tensor * K_prep = ggml_permute(ctx0, K_cur, 0, 2, 1, 3);270                struct ggml_tensor * Q_prep = ggml_permute(ctx0, Q_u,   0, 2, 1, 3);271                struct ggml_tensor * content_scores = ggml_mul_mat(ctx0, K_prep, Q_prep);272                ggml_format_name(content_scores, "enc_%d_attn_content_scores", il);273 274                struct ggml_tensor * Q_v = ggml_add(ctx0, Q_cur, layer.pos_bias_v);275                ggml_format_name(Q_v, "enc_%d_attn_q_v", il);276 277                Q_v = ggml_permute(ctx0, Q_v, 0, 2, 1, 3);278                Q_v = ggml_cont(ctx0, Q_v);279                ggml_format_name(Q_v, "enc_%d_attn_q_v_perm", il);280 281                struct ggml_tensor * rel_pos_scores = ggml_mul_mat(ctx0, pos, Q_v);282                ggml_format_name(rel_pos_scores, "enc_%d_attn_rel_pos", il);283 284                // Relative positional shift285                {286                    const auto pos_window = rel_pos_scores->ne[0];287                    const auto n_frame    = rel_pos_scores->ne[1];288                    const auto n_head     = rel_pos_scores->ne[2];289 290                    rel_pos_scores = ggml_pad(ctx0, rel_pos_scores, 1, 0, 0, 0);291                    rel_pos_scores = ggml_roll(ctx0, rel_pos_scores, 1, 0, 0, 0);292 293                    rel_pos_scores = ggml_reshape_3d(ctx0, rel_pos_scores, n_frame, pos_window + 1, n_head);294                    rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);295                    ggml_format_name(rel_pos_scores, "enc_%d_attn_rel_pos_reshaped", il);296 297                    int center = pos_window / 2;298                    size_t offset = rel_pos_scores->nb[0] * (center+1);299 300                    rel_pos_scores = ggml_view_3d(ctx0, rel_pos_scores,301                                                  n_frame, pos_window, n_head,302                                                  (pos_window) * 4,303                                                  rel_pos_scores->nb[2],304                                                  offset);305                    rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);306                    ggml_format_name(rel_pos_scores, "enc_%d_attn_rel_pos_shifted", il);307 308                    rel_pos_scores = ggml_view_3d(ctx0, rel_pos_scores,309                                                  content_scores->ne[0],310                                                  content_scores->ne[1],311                                                  rel_pos_scores->ne[2],312                                                  rel_pos_scores->nb[1],313                                                  rel_pos_scores->nb[2],314                                                  0);315                    rel_pos_scores = ggml_cont(ctx0, rel_pos_scores);316                    ggml_format_name(rel_pos_scores, "enc_%d_attn_rel_pos_shifted_view", il);317                }318 319                struct ggml_tensor * attn_scores = ggml_add(ctx0, content_scores, rel_pos_scores);320                ggml_format_name(attn_scores, "enc_%d_attn_scores", il);321                attn_scores = ggml_scale(ctx0, attn_scores, 1.0f / std::sqrt(d_head));322                attn_scores = ggml_add(ctx0, attn_scores, attn_mask);323                ggml_format_name(attn_scores, "enc_%d_attn_scores_scaled", il);324 325                struct ggml_tensor * probs = ggml_soft_max(ctx0, attn_scores);326                ggml_format_name(probs, "enc_%d_attn_probs", il);327 328                V_cur = ggml_cont(ctx0, ggml_permute(ctx0, V_cur, 1, 2, 0, 3));329                ggml_format_name(V_cur, "enc_%d_attn_v_cur", il);330                cur = ggml_mul_mat(ctx0, probs, V_cur);331                ggml_format_name(cur, "enc_%d_attn_inp", il);332 333                cur = ggml_permute(ctx0, cur, 2, 0, 1, 3);334                cur = ggml_cont_2d(ctx0, cur, n_state, n_time);335                cur = build_mm(layer.o_w, cur);336            }337            ggml_format_name(cur, "enc_%d_attn_out", il);338 339            cur = ggml_add(ctx0, residual, cur);340            ggml_format_name(cur, "enc_%d_attn_res", il);341        }342 343        // Convolution344        {345            struct ggml_tensor * residual = cur;346            ggml_format_name(cur, "enc_%d_residual_conv", il);347 348            cur = ggml_norm(ctx0, cur, hparams.eps);349            cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.norm_conv_w), layer.norm_conv_b);350            ggml_format_name(cur, "enc_%d_norm_conv", il);351 352            // pointwise 1d convolution:353            cur = build_mm(layer.conv_pw1_w, cur);354            ggml_format_name(cur, "enc_%d_conv_pw1", il);355 356            {357                int64_t d = cur->ne[0] / 2;358                struct ggml_tensor * signal = ggml_view_2d(ctx0, cur, d, cur->ne[1], cur->nb[1], 0);359                struct ggml_tensor * gate   = ggml_view_2d(ctx0, cur, d, cur->ne[1], cur->nb[1], d * cur->nb[0]);360 361                cur = ggml_mul(ctx0, signal, ggml_sigmoid(ctx0, gate));362                ggml_format_name(cur, "enc_%d_conv_glu", il);363            }364 365            cur = ggml_cont(ctx0, ggml_transpose(ctx0, cur));366 367            // use ggml_ssm_conv for f32 precision368            const int dw_pad = (hparams.audio_conv_kernel_size - 1) / 2;369            cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);370            cur = ggml_roll(ctx0, cur, dw_pad, 0, 0, 0);371            cur = ggml_pad(ctx0, cur, dw_pad, 0, 0, 0);372            ggml_format_name(cur, "enc_%d_conv_dw_pad", il);373 374            cur = ggml_ssm_conv(ctx0, cur, layer.conv_dw_w);375            ggml_format_name(cur, "enc_%d_conv_1d_dw", il);376 377            cur = ggml_sub(ctx0, cur, layer.conv_norm_mean);378            struct ggml_tensor * std = ggml_sqrt(ctx0, layer.conv_norm_var);379            cur = ggml_div(ctx0, cur, std);380            cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.conv_norm_w), layer.conv_norm_b);381            ggml_format_name(cur, "enc_%d_conv_bn", il);382 383            cur = ggml_silu(ctx0, cur);384            ggml_format_name(cur, "enc_%d_conv_silu", il);385 386            cur = build_mm(layer.conv_pw2_w, cur);387            ggml_format_name(cur, "enc_%d_conv_pw2", il);388 389            cur = ggml_add(ctx0, residual, cur);390            ggml_format_name(cur, "enc_%d_conv_res", il);391        }392 393        // FFN2394        {395            struct ggml_tensor * residual = cur;396            cur = ggml_norm(ctx0, cur, hparams.eps);397            cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.ff_norm_1_w), layer.ff_norm_1_b);398            ggml_format_name(cur, "enc_%d_ffn_norm_2", il);399 400            cur = build_ffn(cur, layer.ff_up_1_w, nullptr, nullptr, nullptr, layer.ff_down_1_w, nullptr, FFN_SILU, il);401            cur = ggml_add(ctx0, residual, ggml_scale(ctx0, cur, 0.5));402            ggml_format_name(cur, "enc_%d_ffn_res", il);403        }404 405        cur = ggml_norm(ctx0, cur, hparams.eps);406        cur = ggml_add(ctx0, ggml_mul(ctx0, cur, layer.ln_2_w), layer.ln_2_b);407    }408 409    cb(cur, "encoder_out", -1);410 411    cur = ggml_rms_norm(ctx0, cur, 1e-6);412    cur = ggml_mul(ctx0, cur, model.mm_norm_pre_w);413    cb(cur, "sound_projection.norm", -1);414 415    cur = build_ffn(cur, model.mm_0_w, model.mm_0_b, nullptr, nullptr, model.mm_1_w, model.mm_1_b, FFN_RELU_SQR, -1);416    cb(cur, "projected", -1);417 418    ggml_build_forward_expand(gf, cur);419 420    return gf;421}422