Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
qwen3tts-gen.cpp767 linesDownload Raw Back to models
1#include "models.h"2 3#include <string>4 5// on-device sampling: top-k, top-p, then a random draw6ggml_tensor * clip_graph_qwen3tts_gen::code_gen::do_sampling(ggml_tensor * logits, ggml_tensor * inp_rand) const {7    logits = ggml_reshape_1d(ctx0, logits, ggml_nelements(logits));8    const int64_t n_vocab = logits->ne[0];9 10    // sort a's rows by idx11    auto sort_by = [this](ggml_tensor * a, ggml_tensor * idx) {12        ggml_tensor * a2d = ggml_reshape_2d(ctx0, a, 1, a->ne[0]);13        return ggml_reshape_1d(ctx0, ggml_get_rows(ctx0, a2d, idx), idx->ne[0]);14    };15 16    ggml_tensor * cur        = logits;17    ggml_tensor * candidates = nullptr; // maps row index back to vocab id18 19    if (top_k > 0 && top_k < n_vocab) {20        ggml_tensor * idx = ggml_top_k(ctx0, cur, top_k);21        candidates = idx;22        cur        = sort_by(cur, idx);23        cb(cur, "sample_top_k_logits", -1);24    }25 26    if (top_p < 1.0f) {27        ggml_tensor * sorted_idx    = ggml_argsort(ctx0, cur, GGML_SORT_ORDER_DESC);28        ggml_tensor * sorted_logits = sort_by(cur, sorted_idx);29        candidates = candidates ? sort_by(candidates, sorted_idx) : sorted_idx;30 31        ggml_tensor * probs = ggml_soft_max(ctx0, sorted_logits);32        ggml_tensor * cdf   = ggml_cumsum(ctx0, probs);33 34        // keep_mask[i] = 1 once cdf[i] crosses top_p35        ggml_tensor * cdf_scaled = ggml_scale_bias(ctx0, cdf, -1.0f, top_p);36        ggml_tensor * keep_mask  = ggml_step(ctx0, cdf_scaled);37        ggml_tensor * idxf       = ggml_sum(ctx0, keep_mask);38        idxf = ggml_clamp(ctx0, idxf, 0.0f, (float) keep_mask->ne[0] - 1);39        ggml_tensor * ones = ggml_scale_bias(ctx0, idxf, 0.0f, 1.0f);40 41        // top-p must include the crossing element, so force it to 142        ggml_tensor * keep_mask_2d = ggml_reshape_2d(ctx0, keep_mask, 1, keep_mask->ne[0]);43        keep_mask_2d = ggml_set_rows(ctx0, keep_mask_2d, ones, ggml_cast(ctx0, idxf, GGML_TYPE_I32));44        keep_mask    = ggml_reshape_1d(ctx0, keep_mask_2d, keep_mask->ne[0]);45 46        // log(1) = 0 (keep), log(0) = -inf (drop)47        ggml_tensor * bias = ggml_log(ctx0, keep_mask);48        cur = ggml_add(ctx0, sorted_logits, bias);49        cb(cur, "sample_top_p_logits", -1);50    }51 52    // draw one token: find where the cdf crosses inp_rand53    ggml_tensor * probs  = ggml_soft_max(ctx0, cur);54    ggml_tensor * cumsum = ggml_cumsum(ctx0, probs);55 56    ggml_tensor * diff       = ggml_sub(ctx0, cumsum, inp_rand);57    ggml_tensor * cross_mask = ggml_step(ctx0, diff);58    ggml_tensor * idxf       = ggml_sum(ctx0, cross_mask);59    ggml_tensor * idx        = ggml_cast(ctx0, ggml_scale_bias(ctx0, idxf, -1.0f, (float) cross_mask->ne[0]), GGML_TYPE_I32);60 61    if (candidates) {62        ggml_tensor * cand_2d = ggml_reshape_2d(ctx0, candidates, 1, candidates->ne[0]);63        idx = ggml_get_rows(ctx0, cand_2d, idx);64    }65    cb(idx, "sample_token_id", -1);66 67    return idx;68}69 70// returns a new cache with row row_idx set to value71ggml_tensor * clip_graph_qwen3tts_gen::code_gen::cache_set(ggml_tensor * cache, int row_idx, ggml_tensor * value) const {72    const int64_t n_embd  = cache->ne[0];73    const int64_t n_cache = cache->ne[1];74    GGML_ASSERT(row_idx >= 0 && row_idx < n_cache);75 76    // append value as the last row, then gather it back into place77    ggml_tensor * value_2d  = ggml_reshape_2d(ctx0, value, n_embd, 1);78    ggml_tensor * cache_ext = ggml_concat(ctx0, cache, value_2d, 1); // [n_embd, n_cache + 1]79 80    // gather indices [0..row_idx-1, n_cache, row_idx+1..n_cache-1]81    // built via concat, since ggml_set_rows needs F32/F16 values, not an I32 index array82    ggml_tensor * idx = const_i32(cache, (float) n_cache);83    if (row_idx > 0) {84        ggml_tensor * prefix = ggml_cast(ctx0, ggml_arange(ctx0, 0.0f, (float) row_idx, 1.0f), GGML_TYPE_I32);85        idx = ggml_concat(ctx0, prefix, idx, 0);86    }87    if (row_idx < n_cache - 1) {88        ggml_tensor * suffix = ggml_cast(ctx0, ggml_arange(ctx0, (float) (row_idx + 1), (float) n_cache, 1.0f), GGML_TYPE_I32);89        idx = ggml_concat(ctx0, idx, suffix, 0);90    }91 92    ggml_tensor * result = ggml_get_rows(ctx0, cache_ext, idx);93    cb(result, "cache_set_out", -1);94    return result;95}96 97// builds a const i32 with no host upload: view a tensor, zero it via scale, add value, cast to i3298ggml_tensor * clip_graph_qwen3tts_gen::code_gen::const_i32(ggml_tensor * anchor, float value) const {99    ggml_tensor * v = ggml_view_1d(ctx0, anchor, 1, 0);100    if (v->type != GGML_TYPE_F32) {101        v = ggml_cast(ctx0, v, GGML_TYPE_F32);102    }103    return ggml_cast(ctx0, ggml_scale_bias(ctx0, v, 0.0f, value), GGML_TYPE_I32);104}105 106// causal keep-mask row for a query at position pos, window size n_kv_pad107ggml_tensor * clip_graph_qwen3tts_gen::code_gen::causal_mask_row(int64_t n_kv_pad, int pos) const {108    ggml_tensor * ones = ggml_fill(ctx0, ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, n_kv_pad, n_kv_pad), 1.0f);109    ggml_tensor * keep = ggml_tri(ctx0, ones, GGML_TRI_TYPE_LOWER_DIAG);110    ggml_tensor * row  = ggml_view_1d(ctx0, keep, n_kv_pad, (size_t) pos * keep->nb[1]);111    ggml_tensor * mask = ggml_log(ctx0, row); // 0 = keep, -inf = masked112    return ggml_reshape_4d(ctx0, mask, n_kv_pad, 1, 1, 1);113}114 115// talker hidden size -> predictor hidden size (small_to_mtp_projection)116ggml_tensor * clip_graph_qwen3tts_gen::code_gen::project_in(ggml_tensor * cur) const {117    if (!model.gen_code_proj_in_w) {118        return cur;119    }120    cur = ggml_mul_mat(ctx0, model.gen_code_proj_in_w, cur);121    if (model.gen_code_proj_in_b) {122        cur = ggml_add(ctx0, cur, model.gen_code_proj_in_b);123    }124    return cur;125}126 127// one transformer layer at position pos; writes k/v into k_cache_layer/v_cache_layer at row pos128ggml_tensor * clip_graph_qwen3tts_gen::code_gen::layer_forward(129        ggml_tensor * cur,130        const clip_layer & layer,131        ggml_tensor * inp_pos,132        ggml_tensor * kq_mask,133        ggml_tensor *& k_cache_layer,134        ggml_tensor *& v_cache_layer,135        int64_t n_kv_pad,136        int pos,137        int il) const {138    const int     n_head    = hparams.n_head;139    const int     n_head_kv = hparams.n_head_kv;140    const int64_t d_head    = layer.q_w->ne[1] / n_head; // real head_dim, not n_embd / n_head141    const float   kq_scale  = 1.0f / sqrtf((float) d_head);142 143    ggml_tensor * residual = cur;144 145    ggml_tensor * h = ggml_rms_norm(ctx0, cur, hparams.eps);146    h = ggml_mul(ctx0, h, layer.ln_1_w);147 148    ggml_tensor * q = ggml_mul_mat(ctx0, layer.q_w, h);149    ggml_tensor * k = ggml_mul_mat(ctx0, layer.k_w, h);150    ggml_tensor * v = ggml_mul_mat(ctx0, layer.v_w, h);151 152    q = ggml_reshape_3d(ctx0, q, d_head, n_head, 1);153    k = ggml_reshape_3d(ctx0, k, d_head, n_head_kv, 1);154 155    q = ggml_rms_norm(ctx0, q, hparams.eps);156    q = ggml_mul(ctx0, q, layer.q_norm);157    k = ggml_rms_norm(ctx0, k, hparams.eps);158    k = ggml_mul(ctx0, k, layer.k_norm);159 160    q = ggml_rope_ext(ctx0, q, inp_pos, nullptr, (int) d_head, GGML_ROPE_TYPE_NEOX, 0,161                      hparams.rope_theta, 1.0f, 0.0f, 1.0f, 0.0f, 0.0f);162    k = ggml_rope_ext(ctx0, k, inp_pos, nullptr, (int) d_head, GGML_ROPE_TYPE_NEOX, 0,163                      hparams.rope_theta, 1.0f, 0.0f, 1.0f, 0.0f, 0.0f);164 165    // write k/v into the cache at row pos, flat layout166    ggml_tensor * k_flat = ggml_reshape_1d(ctx0, k, d_head * n_head_kv);167    k_cache_layer = cache_set(k_cache_layer, pos, k_flat);168    v_cache_layer = cache_set(v_cache_layer, pos, v);169 170    ggml_tensor * q_cur = ggml_reshape_4d(ctx0, q, d_head, n_head, 1, 1);171    ggml_tensor * k_cur = ggml_reshape_4d(ctx0, k_cache_layer, d_head, n_head_kv, n_kv_pad, 1);172    ggml_tensor * v_cur = ggml_reshape_4d(ctx0, v_cache_layer, d_head, n_head_kv, n_kv_pad, 1);173 174    ggml_tensor * attn_out = build_attn(layer.o_w, layer.o_b, q_cur, k_cur, v_cur, kq_mask, kq_scale, il);175 176    cur = ggml_add(ctx0, residual, attn_out);177 178    ggml_tensor * h2 = ggml_rms_norm(ctx0, cur, hparams.eps);179    h2 = ggml_mul(ctx0, h2, layer.ln_2_w);180 181    ggml_tensor * gate = ggml_mul_mat(ctx0, layer.ff_gate_w, h2);182    ggml_tensor * up   = ggml_mul_mat(ctx0, layer.ff_up_w, h2);183    ggml_tensor * gu   = ggml_swiglu_split(ctx0, gate, up);184    ggml_tensor * down = ggml_mul_mat(ctx0, layer.ff_down_w, gu);185 186    return ggml_add(ctx0, cur, down);187}188 189// position 0: hidden bridge, seeds the k/v cache, no sampling190// position 1: embed(code0), sample with lm_head[0], write out_code_cache[1]191void clip_graph_qwen3tts_gen::code_gen::prefill(192        std::vector<ggml_tensor *> & k_cache,193        std::vector<ggml_tensor *> & v_cache,194        ggml_tensor *& out_code_cache,195        ggml_tensor * h_state,196        ggml_tensor * code0_embd,197        ggml_tensor * inp_rand) const {198    const int64_t n_kv_pad = k_cache[0]->ne[1];199 200    {201        ggml_tensor * cur     = project_in(h_state);202        ggml_tensor * kq_mask = causal_mask_row(n_kv_pad, 0);203        ggml_tensor * inp_pos = const_i32(k_cache[0], 0.0f);204        for (size_t il = 0; il < model.layers.size(); il++) {205            cur = layer_forward(cur, model.layers[il], inp_pos, kq_mask, k_cache[il], v_cache[il], n_kv_pad, 0, (int) il);206        }207        // position 0's output is unused, it only seeded the cache208    }209 210    {211        ggml_tensor * cur     = project_in(code0_embd);212        ggml_tensor * kq_mask = causal_mask_row(n_kv_pad, 1);213        ggml_tensor * inp_pos = const_i32(k_cache[0], 1.0f);214        for (size_t il = 0; il < model.layers.size(); il++) {215            cur = layer_forward(cur, model.layers[il], inp_pos, kq_mask, k_cache[il], v_cache[il], n_kv_pad, 1, (int) il);216        }217 218        cur = ggml_rms_norm(ctx0, cur, hparams.eps);219        cur = ggml_mul(ctx0, cur, model.gen_code_norm_w);220 221        ggml_tensor * head_w = model.gen_code_head_w;222        ggml_tensor * head_g = ggml_view_2d(ctx0, head_w, head_w->ne[0], head_w->ne[1], head_w->nb[1], 0); // lm_head[0]223        ggml_tensor * logits = ggml_mul_mat(ctx0, head_g, cur);224 225        ggml_tensor * sampled = do_sampling(logits, inp_rand);226        out_code_cache = cache_set(out_code_cache, 1, sampled);227    }228}229 230// one decode step of code_predictor231// at step_idx g:232// - read code from out_code_cache[g], then embed it with codebook table g-1233// - write new kv at cache row g+1, sample with lm_head[g]234// - write result to out_code_cache[g+1]235// step_idx must be in [1, n_acoustic - 1]236ggml_tensor * clip_graph_qwen3tts_gen::code_gen::step(237        std::vector<ggml_tensor *> & k_cache,238        std::vector<ggml_tensor *> & v_cache,239        ggml_tensor * out_code_cache,240        ggml_tensor * inp_rand,241        int step_idx) const {242    const int64_t n_acoustic = model.gen_code_head_w->ne[2];243    GGML_ASSERT(step_idx >= 1 && step_idx < n_acoustic);244    GGML_ASSERT(k_cache.size() == model.layers.size());245    GGML_ASSERT(v_cache.size() == model.layers.size());246 247    const int64_t n_kv_pad = k_cache[0]->ne[1];248    const int     pos      = step_idx + 1; // new cache row and RoPE position249 250    // embed the previous code via this step's codebook table (rows are already scalars)251    ggml_tensor * code_in = ggml_view_1d(ctx0, out_code_cache, 1, (size_t) step_idx * out_code_cache->nb[1]);252 253    ggml_tensor * embd_w = model.gen_code_embd_w; // [n_embd_talker, vocab, n_acoustic]254    ggml_tensor * embd_g = ggml_view_2d(ctx0, embd_w, embd_w->ne[0], embd_w->ne[1], embd_w->nb[1],255                                        (size_t) (step_idx - 1) * embd_w->nb[2]);256    ggml_tensor * cur = ggml_get_rows(ctx0, embd_g, code_in);257    cur = ggml_reshape_1d(ctx0, cur, cur->ne[0]);258    cb(cur, "step_embd_in", step_idx);259 260    cur = project_in(cur);261    cb(cur, "step_proj_in", step_idx);262 263    ggml_tensor * kq_mask = causal_mask_row(n_kv_pad, pos);264    ggml_tensor * inp_pos = const_i32(k_cache[0], (float) pos);265 266    for (size_t il = 0; il < model.layers.size(); il++) {267        cur = layer_forward(cur, model.layers[il], inp_pos, kq_mask, k_cache[il], v_cache[il], n_kv_pad, pos, (int) il);268        cb(cur, "step_layer_out", (int) il);269    }270 271    // final norm, this step's lm_head, sample, write the result272    cur = ggml_rms_norm(ctx0, cur, hparams.eps);273    cur = ggml_mul(ctx0, cur, model.gen_code_norm_w);274 275    ggml_tensor * head_w = model.gen_code_head_w; // [n_embd_pred, vocab, n_acoustic]276    ggml_tensor * head_g = ggml_view_2d(ctx0, head_w, head_w->ne[0], head_w->ne[1], head_w->nb[1],277                                        (size_t) step_idx * head_w->nb[2]);278    ggml_tensor * logits = ggml_mul_mat(ctx0, head_g, cur);279    cb(logits, "step_logits", step_idx);280 281    ggml_tensor * sampled = do_sampling(logits, inp_rand);282    cb(sampled, "step_sampled", step_idx);283 284    return cache_set(out_code_cache, pos, sampled);285}286 287// causal conv1d, stride 1: prepend persisted left-context instead of zero-padding, then a plain conv288// x: [T, IC] (T-first). w: [K, IC, OC]. state_name empty means K == 1 (no left-context). returns [T, OC]289ggml_tensor * clip_graph_qwen3tts_gen::code2wav::causal_conv1d(ggml_tensor * x, ggml_tensor * w, ggml_tensor * b, int dilation, const std::string & state_name) const {290    const int K   = (int) w->ne[0];291    const int pad = (K - 1) * dilation;292 293    ggml_tensor * x_full = x;294    if (pad > 0) {295        ggml_tensor * left = state_in.at(state_name); // [pad, IC]296        x_full = ggml_concat(ctx0, left, x, 0);297    }298    ggml_tensor * y = ggml_conv_1d(ctx0, w, x_full, 1, 0, dilation); // [T, OC, 1]299    y = ggml_reshape_2d(ctx0, y, y->ne[0], y->ne[1]);300    if (b) {301        y = ggml_add(ctx0, y, ggml_reshape_2d(ctx0, b, 1, b->ne[0]));302    }303    if (pad > 0) {304        ggml_tensor * new_left = ggml_cont(ctx0, ggml_view_2d(ctx0, x_full, pad, x_full->ne[1], x_full->nb[1],305                                                              (size_t) (x_full->ne[0] - pad) * x_full->nb[0]));306        state_out.push_back({state_name, new_left});307    }308    return y;309}310 311// causal depthwise conv1d, stride 1, dilation 1, kernel from w's shape.312// x: [T, C]. w: [K, 1, C]. returns [T, C]. see causal_conv1d for the state contract.313ggml_tensor * clip_graph_qwen3tts_gen::code2wav::causal_conv1d_dw(ggml_tensor * x, ggml_tensor * w, ggml_tensor * b, const std::string & state_name) const {314    const int K   = (int) w->ne[0];315    const int pad = K - 1;316 317    ggml_tensor * x_full = x;318    if (pad > 0) {319        ggml_tensor * left = state_in.at(state_name); // [pad, C]320        x_full = ggml_concat(ctx0, left, x, 0);321    }322    ggml_tensor * y = ggml_conv_1d_dw(ctx0, w, x_full, 1, 0, 1); // [T, C, 1]323    y = ggml_reshape_2d(ctx0, y, y->ne[0], y->ne[1]);324    if (b) {325        y = ggml_add(ctx0, y, ggml_reshape_2d(ctx0, b, 1, b->ne[0]));326    }327    if (pad > 0) {328        ggml_tensor * new_left = ggml_cont(ctx0, ggml_view_2d(ctx0, x_full, pad, x_full->ne[1], x_full->nb[1],329                                                              (size_t) (x_full->ne[0] - pad) * x_full->nb[0]));330        state_out.push_back({state_name, new_left});331    }332    return y;333}334 335// causal ConvTranspose1d, the (kernel - stride) overlap tail is kept as state for the next call336// x: [T, IC], w: [K, OC, IC]. state_name empty means K == stride (no overlap). returns [T * stride, OC]337ggml_tensor * clip_graph_qwen3tts_gen::code2wav::causal_conv_transpose1d(ggml_tensor * x, ggml_tensor * w, ggml_tensor * b, int stride, const std::string & state_name) const {338    const int     K        = (int) w->ne[0];339    const int     OC       = (int) w->ne[1];340    const int     trim     = K - stride;341    const int64_t emit_len = x->ne[0] * stride;342 343    // transposed conv as GEMM + col2im scatter-add, y: [emit_len + trim, OC]344    ggml_tensor * w2  = ggml_reshape_2d(ctx0, w, (int64_t) K * OC, w->ne[2]);345    w2                = ggml_cont(ctx0, ggml_transpose(ctx0, w2));346    ggml_tensor * xt  = ggml_cont(ctx0, ggml_transpose(ctx0, x));347    ggml_tensor * col = ggml_mul_mat(ctx0, w2, xt);348    ggml_tensor * y   = ggml_col2im_1d(ctx0, col, stride, OC, 0);349 350    ggml_tensor * out = y;351    if (trim > 0) {352        ggml_tensor * tail = state_in.at(state_name); // [trim, OC]353        ggml_tensor * head = ggml_add(ctx0, ggml_view_2d(ctx0, y, trim, y->ne[1], y->nb[1], 0), tail);354        if (emit_len > trim) {355            ggml_tensor * middle = ggml_view_2d(ctx0, y, emit_len - trim, y->ne[1], y->nb[1], (size_t) trim * y->nb[0]);356            out = ggml_concat(ctx0, head, middle, 0);357        } else {358            out = head;359        }360        ggml_tensor * new_tail = ggml_cont(ctx0, ggml_view_2d(ctx0, y, trim, y->ne[1], y->nb[1], (size_t) emit_len * y->nb[0]));361        state_out.push_back({state_name, new_tail});362    }363    if (b) {364        out = ggml_add(ctx0, out, ggml_reshape_2d(ctx0, b, 1, b->ne[0]));365    }366    return out;367}368 369// SnakeBeta activation: y = x + sin(alpha*x)^2 * inv_beta (alpha/inv_beta folded via exp/reciprocal at conversion time)370// x: [T, C]. alpha/beta: [C], broadcasts over T371ggml_tensor * clip_graph_qwen3tts_gen::code2wav::snake(ggml_tensor * x, ggml_tensor * alpha, ggml_tensor * beta) const {372    ggml_tensor * a = ggml_reshape_2d(ctx0, alpha, 1, alpha->ne[0]);373    ggml_tensor * b = ggml_reshape_2d(ctx0, beta,  1, beta->ne[0]);374 375    // expand reshapes first so mul/sin/sqr/mul/add lands as consecutive nodes, letting backends fuse them376    ggml_build_forward_expand(gf, a);377    ggml_build_forward_expand(gf, b);378 379    ggml_tensor * s = ggml_sin(ctx0, ggml_mul(ctx0, x, a));380    s = ggml_sqr(ctx0, s);381    s = ggml_mul(ctx0, s, b);382    return ggml_add(ctx0, x, s);383}384 385// RVQ codebook decode: T frames of 16 codes -> 512-dim hidden (C-first, [512, T])386// codebook 0 (semantic) and 1..15 (acoustic) sum within their group, project separately, then add387ggml_tensor * clip_graph_qwen3tts_gen::code2wav::quant_decode(ggml_tensor * inp_codes) const {388    const auto & c2w = model.c2w;389    const int64_t T = inp_codes->ne[0];390 391    // ids for codebook group g over all T frames, [T] I32392    auto group_ids = [&](int g) {393        return ggml_view_1d(ctx0, inp_codes, T, (size_t) g * inp_codes->nb[1]);394    };395 396    ggml_tensor * sem     = ggml_get_rows(ctx0, c2w.quant_first_cb_w, group_ids(0)); // [256, T]397    ggml_tensor * sem_out = ggml_mul_mat(ctx0, c2w.quant_first_out_w, sem);          // [512, T]398 399    ggml_tensor * acc = nullptr;400    const int64_t n_acoustic = c2w.quant_rest_cb_w->ne[2];401    for (int g = 1; g <= n_acoustic; g++) {402        ggml_tensor * cb_g  = ggml_view_2d(ctx0, c2w.quant_rest_cb_w, c2w.quant_rest_cb_w->ne[0], c2w.quant_rest_cb_w->ne[1],403                                           c2w.quant_rest_cb_w->nb[1], (size_t) (g - 1) * c2w.quant_rest_cb_w->nb[2]);404        ggml_tensor * embd = ggml_get_rows(ctx0, cb_g, group_ids(g)); // [256, T]405        acc = acc ? ggml_add(ctx0, acc, embd) : embd;406    }407    ggml_tensor * ac_out = ggml_mul_mat(ctx0, c2w.quant_rest_out_w, acc); // [512, T]408 409    ggml_tensor * hidden = ggml_add(ctx0, sem_out, ac_out);410    cb(hidden, "wav_quant_hidden", -1);411    return hidden;412}413 414// one pre_transformer layer over a batch of N = sliding_window new frames415// attention runs over [(W-1)-frame prefix from the last batch] + [N new frames]416// RoPE positions come from a persisted counter, so phases line up across batches417ggml_tensor * clip_graph_qwen3tts_gen::code2wav::tfm_layer_forward(ggml_tensor * cur, const clip_layer & layer, int il) const {418    const int     n_head    = hparams.wav_tfm_n_head;419    const int     n_head_kv = hparams.wav_tfm_n_head_kv;420    const int64_t d_head    = layer.q_w->ne[1] / n_head;421    const float   kq_scale  = 1.0f / sqrtf((float) d_head);422    const int64_t W         = hparams.wav_tfm_swa; // == N, frames per batch423    const int64_t N         = cur->ne[1];424    const int64_t prefix    = W - 1;425    const int64_t total_kv  = prefix + N;426 427    ggml_tensor * residual = cur;428    ggml_tensor * h = ggml_rms_norm(ctx0, cur, hparams.wav_tfm_eps);429    h = ggml_mul(ctx0, h, layer.ln_1_w);430 431    ggml_tensor * q = ggml_mul_mat(ctx0, layer.q_w, h); // [n_head*d_head, N]432    ggml_tensor * k = ggml_mul_mat(ctx0, layer.k_w, h); // [n_head_kv*d_head, N]433    ggml_tensor * v = ggml_mul_mat(ctx0, layer.v_w, h); // [n_head_kv*d_head, N]434 435    q = ggml_reshape_3d(ctx0, q, d_head, n_head, N);436    k = ggml_reshape_3d(ctx0, k, d_head, n_head_kv, N);437 438    // real, ever-increasing positions: base (persisted) .. base+N-1439    ggml_tensor * base   = ggml_reshape_1d(ctx0, state_in.at("tfm_pos"), 1);440    ggml_tensor * offset = ggml_arange(ctx0, 0.0f, (float) N, 1.0f);441    ggml_tensor * pos    = ggml_cast(ctx0, ggml_add(ctx0, offset, base), GGML_TYPE_I32);442 443    q = ggml_rope_ext(ctx0, q, pos, nullptr, (int) d_head, GGML_ROPE_TYPE_NEOX, 0,444                      hparams.wav_tfm_rope_theta, 1.0f, 0.0f, 1.0f, 0.0f, 0.0f);445    k = ggml_rope_ext(ctx0, k, pos, nullptr, (int) d_head, GGML_ROPE_TYPE_NEOX, 0,446                      hparams.wav_tfm_rope_theta, 1.0f, 0.0f, 1.0f, 0.0f, 0.0f);447 448    // the position counter is the same for all layers, push it once from layer 0449    if (il == 0) {450        state_out.push_back({"tfm_pos", ggml_scale_bias(ctx0, state_in.at("tfm_pos"), 1.0f, (float) N)});451    }452 453    ggml_tensor * k_new = ggml_reshape_2d(ctx0, k, d_head * n_head_kv, N);454    ggml_tensor * v_new = ggml_reshape_2d(ctx0, v, d_head * n_head_kv, N);455 456    ggml_tensor * old_k = state_in.at("tfm_k_" + std::to_string(il)); // [d_head*n_head_kv, W-1]457    ggml_tensor * old_v = state_in.at("tfm_v_" + std::to_string(il));458 459    ggml_tensor * k_full = ggml_concat(ctx0, old_k, k_new, 1); // [.., prefix+N]460    ggml_tensor * v_full = ggml_concat(ctx0, old_v, v_new, 1);461 462    // next batch's prefix: the last (W-1) frames of this batch463    state_out.push_back({"tfm_k_" + std::to_string(il),464        ggml_cont(ctx0, ggml_view_2d(ctx0, k_full, k_full->ne[0], prefix, k_full->nb[1], (size_t) N * k_full->nb[1]))});465    state_out.push_back({"tfm_v_" + std::to_string(il),466        ggml_cont(ctx0, ggml_view_2d(ctx0, v_full, v_full->ne[0], prefix, v_full->nb[1], (size_t) N * v_full->nb[1]))});467 468    // banded causal mask: key j is visible to query i iff 0 <= (prefix+i) - j < W469    ggml_tensor * pos_k = ggml_reshape_2d(ctx0, ggml_arange(ctx0, 0.0f, (float) total_kv, 1.0f), total_kv, 1);470    ggml_tensor * pos_q = ggml_reshape_2d(ctx0, ggml_arange(ctx0, (float) prefix, (float) (prefix + N), 1.0f), 1, N);471    ggml_tensor * pos_q_grid = ggml_repeat_4d(ctx0, pos_q, total_kv, N, 1, 1);472    ggml_tensor * diff = ggml_sub(ctx0, pos_q_grid, pos_k); // [total_kv, N]473 474    ggml_tensor * causal_keep = ggml_step(ctx0, ggml_scale_bias(ctx0, diff, 1.0f, 0.5f));            // diff >= 0475    ggml_tensor * in_window   = ggml_step(ctx0, ggml_scale_bias(ctx0, diff, -1.0f, (float) W - 0.5f)); // diff < W476    ggml_tensor * keep = ggml_mul(ctx0, causal_keep, in_window);477 478    // on a cold start, key j is real state only when j >= prefix - tfm_pos, mask out the rest479    ggml_tensor * warm = ggml_step(ctx0, ggml_scale_bias(ctx0, ggml_add(ctx0, pos_k, base),480                                                         1.0f, 0.5f - (float) prefix)); // j + pos > prefix - 0.5481    keep = ggml_mul(ctx0, keep, warm);482 483    ggml_tensor * mask = ggml_reshape_4d(ctx0, ggml_log(ctx0, keep), total_kv, N, 1, 1); // 0 = keep, -inf = masked484 485    ggml_tensor * q_cur = ggml_reshape_4d(ctx0, q, d_head, n_head, N, 1);486    ggml_tensor * k_cur = ggml_reshape_4d(ctx0, k_full, d_head, n_head_kv, total_kv, 1);487    ggml_tensor * v_cur = ggml_reshape_4d(ctx0, v_full, d_head, n_head_kv, total_kv, 1);488 489    ggml_tensor * attn_out = build_attn(layer.o_w, layer.o_b, q_cur, k_cur, v_cur, mask, kq_scale, il);490    if (layer.ls_1_w) {491        attn_out = ggml_mul(ctx0, attn_out, layer.ls_1_w);492    }493    cur = ggml_add(ctx0, residual, attn_out);494 495    ggml_tensor * residual2 = cur;496    ggml_tensor * h2 = ggml_rms_norm(ctx0, cur, hparams.wav_tfm_eps);497    h2 = ggml_mul(ctx0, h2, layer.ln_2_w);498 499    ggml_tensor * gate = ggml_mul_mat(ctx0, layer.ff_gate_w, h2);500    ggml_tensor * up   = ggml_mul_mat(ctx0, layer.ff_up_w, h2);501    ggml_tensor * gu   = ggml_swiglu_split(ctx0, gate, up);502    ggml_tensor * down = ggml_mul_mat(ctx0, layer.ff_down_w, gu);503    if (layer.ls_2_w) {504        down = ggml_mul(ctx0, down, layer.ls_2_w);505    }506    return ggml_add(ctx0, residual2, down);507}508 509// dwconv -> LayerNorm -> pwconv1 -> GELU -> pwconv2 -> layer scale -> residual510// x: [T, C] T-first; LayerNorm/pwconv need C on ne0, so this transposes in and back out511ggml_tensor * clip_graph_qwen3tts_gen::code2wav::convnext_block(ggml_tensor * x, const clip_code2wav::upsample_block & blk, const std::string & state_prefix) const {512    ggml_tensor * residual = x;513 514    ggml_tensor * h = causal_conv1d_dw(x, blk.dwconv_w, blk.dwconv_b, state_prefix + "_dwconv"); // [T, C]515    ggml_tensor * hc = ggml_cont(ctx0, ggml_transpose(ctx0, h)); // [C, T]516 517    hc = ggml_norm(ctx0, hc, 1e-6f);518    hc = ggml_mul(ctx0, hc, blk.norm_w);519    hc = ggml_add(ctx0, hc, blk.norm_b);520 521    ggml_tensor * g = ggml_mul_mat(ctx0, blk.pw1_w, hc);522    g = ggml_add(ctx0, g, blk.pw1_b);523    g = ggml_gelu(ctx0, g);524    g = ggml_mul_mat(ctx0, blk.pw2_w, g);525    g = ggml_add(ctx0, g, blk.pw2_b);526    g = ggml_mul(ctx0, g, blk.gamma);527 528    ggml_tensor * g_t = ggml_cont(ctx0, ggml_transpose(ctx0, g)); // back to [T, C]529    return ggml_add(ctx0, residual, g_t);530}531 532// SnakeBeta -> dilated causal conv (k=7) -> SnakeBeta -> pointwise causal conv (k=1) -> residual.533// x: [T, C]. returns [T, C].534ggml_tensor * clip_graph_qwen3tts_gen::code2wav::dac_res_unit(ggml_tensor * x, const clip_code2wav::dac_res & res, int dilation, const std::string & state_name) const {535    ggml_tensor * residual = x;536    ggml_tensor * h = snake(x, res.act1_alpha, res.act1_beta);537    h = causal_conv1d(h, res.conv1_w, res.conv1_b, dilation, state_name);538    h = snake(h, res.act2_alpha, res.act2_beta);539    h = causal_conv1d(h, res.conv2_w, res.conv2_b, 1, ""); // k=1, no left-context needed540    return ggml_add(ctx0, residual, h);541}542 543// RVQ codes -> raw PCM for a batch of N = sliding_window frames544ggml_tensor * clip_graph_qwen3tts_gen::code2wav::decode(ggml_tensor * inp_codes) const {545    const auto & c2w = model.c2w;546 547    // 1. quantizer decode: N frames of 16 codes -> [512, N] (C-first)548    ggml_tensor * hidden = quant_decode(inp_codes);549 550    // 2. pre_conv: [512, N] -> T-first [N, 512] -> causal conv k=3 -> [N, 1024]551    ggml_tensor * x = ggml_cont(ctx0, ggml_transpose(ctx0, hidden)); // [N, 512]552    x = causal_conv1d(x, c2w.pre_conv_w, c2w.pre_conv_b, 1, "pre_conv"); // [N, 1024]553    cb(x, "wav_pre_conv_out", -1);554 555    // 3. pre_transformer: back to C-first [1024, N], project down, run the layers, project back up556    ggml_tensor * cur = ggml_cont(ctx0, ggml_transpose(ctx0, x)); // [1024, N]557    cur = ggml_mul_mat(ctx0, c2w.tfm_in_proj_w, cur);558    cur = ggml_add(ctx0, cur, c2w.tfm_in_proj_b); // [512 (tfm hidden), N]559 560    for (int il = 0; il < hparams.wav_tfm_n_layer; il++) {561        cur = tfm_layer_forward(cur, c2w.tfm_layers[il], il);562    }563 564    cur = ggml_rms_norm(ctx0, cur, hparams.wav_tfm_eps);565    cur = ggml_mul(ctx0, cur, c2w.tfm_output_norm_w);566    cur = ggml_mul_mat(ctx0, c2w.tfm_out_proj_w, cur);567    cur = ggml_add(ctx0, cur, c2w.tfm_out_proj_b); // [1024, N]568    cb(cur, "wav_tfm_out", -1);569 570    // 4. upsample: 2x (causal ConvTranspose1d, stride 2 + ConvNeXt block), back to T-first571    // kernel == stride here, so there is no overlap tail to persist572    x = ggml_cont(ctx0, ggml_transpose(ctx0, cur)); // [N, 1024]573    for (size_t il = 0; il < c2w.upsample.size(); il++) {574        const auto & up = c2w.upsample[il];575        x = causal_conv_transpose1d(x, up.conv_w, up.conv_b, 2, "");576        x = convnext_block(x, up, "up" + std::to_string(il));577        cb(x, "wav_upsample_out", (int) il);578    }579 580    // 5. DAC decoder: conv_pre -> n blocks (SnakeBeta -> ConvTranspose1d -> 3 res units) -> conv_post581    static constexpr int DAC_DILATIONS[3] = { 1, 3, 9 };582 583    x = causal_conv1d(x, c2w.dac_entry_w, c2w.dac_entry_b, 1, "dac_entry");584    cb(x, "wav_dac_entry_out", -1);585 586    for (size_t il = 0; il < c2w.dac.size(); il++) {587        const auto & blk = c2w.dac[il];588        const int stride = (int) (blk.conv_w->ne[0] / 2); // kernel == 2*stride for all 4 blocks589        const std::string blk_name = "dac" + std::to_string(il);590        x = snake(x, blk.snake_alpha, blk.snake_beta);591        x = causal_conv_transpose1d(x, blk.conv_w, blk.conv_b, stride, blk_name + "_tail");592        for (size_t ir = 0; ir < blk.res.size(); ir++) {593            x = dac_res_unit(x, blk.res[ir], DAC_DILATIONS[ir], blk_name + "_res" + std::to_string(ir));594        }595        cb(x, "wav_dac_block_out", (int) il);596    }597 598    x = snake(x, c2w.dac_post_snake_alpha, c2w.dac_post_snake_beta);599    x = causal_conv1d(x, c2w.dac_post_conv_w, c2w.dac_post_conv_b, 1, "dac_post_conv"); // [n_samples, 1]600 601    x = ggml_clamp(ctx0, x, -1.0f, 1.0f);602    x = ggml_reshape_1d(ctx0, x, x->ne[0]);603    cb(x, "wav_audio_out", -1);604    return x;605}606 607// code2wav's persisted state buffers: RoPE position counter, K/V per pre_transformer layer,608// left-context/tail per stateful conv. shape lookup only, no graph needed609std::vector<c2w_state_slot> list_c2w_state_slots(const clip_hparams & hparams, const clip_model & model) {610    const auto & c2w = model.c2w;611    std::vector<c2w_state_slot> slots;612 613    slots.push_back({"tfm_pos", 1, 1});614 615    // prefix is (W-1) frames, the batch itself gives the other N=W frames (see tfm_layer_forward)616    const int64_t d_head = c2w.tfm_layers[0].q_w->ne[1] / hparams.wav_tfm_n_head;617    const int64_t kv_ch  = d_head * hparams.wav_tfm_n_head_kv;618    const int64_t prefix = hparams.wav_tfm_swa - 1;619    for (int il = 0; il < hparams.wav_tfm_n_layer; il++) {620        slots.push_back({"tfm_k_" + std::to_string(il), kv_ch, prefix});621        slots.push_back({"tfm_v_" + std::to_string(il), kv_ch, prefix});622    }623 624    slots.push_back({"pre_conv", c2w.pre_conv_w->ne[0] - 1, c2w.pre_conv_w->ne[1]});625 626    for (size_t il = 0; il < c2w.upsample.size(); il++) {627        const auto & up = c2w.upsample[il];628        slots.push_back({"up" + std::to_string(il) + "_dwconv", up.dwconv_w->ne[0] - 1, up.dwconv_w->ne[2]});629    }630 631    slots.push_back({"dac_entry", c2w.dac_entry_w->ne[0] - 1, c2w.dac_entry_w->ne[1]});632 633    static constexpr int DAC_DILATIONS[3] = { 1, 3, 9 };634    for (size_t il = 0; il < c2w.dac.size(); il++) {635        const auto & blk = c2w.dac[il];636        const int64_t stride = blk.conv_w->ne[0] / 2; // kernel == 2*stride for all 4 blocks637        const std::string blk_name = "dac" + std::to_string(il);638        slots.push_back({blk_name + "_tail", stride, blk.conv_w->ne[1]});639        for (size_t ir = 0; ir < blk.res.size(); ir++) {640            const auto & res = blk.res[ir];641            slots.push_back({blk_name + "_res" + std::to_string(ir),642                              (res.conv1_w->ne[0] - 1) * DAC_DILATIONS[ir], res.conv1_w->ne[1]});643        }644    }645 646    slots.push_back({"dac_post_conv", c2w.dac_post_conv_w->ne[0] - 1, c2w.dac_post_conv_w->ne[1]});647 648    return slots;649}650 651// both sub-graphs are always built, so the topology stays constant652// ggml_build_forward_select() then picks the one that actually runs653ggml_cgraph * clip_graph_qwen3tts_gen::build() {654    GGML_ASSERT(n_batch == 1); // this module only ever processes one frame at a time655 656    int idx;657    switch (gen_process) {658        case CLIP_GEN_PROCESS_GEN_CODE: idx = 0; break;659        case CLIP_GEN_PROCESS_GEN_WAV:  idx = 1; break;660        default: GGML_ABORT("unknown gen_process");661    }662 663    // ---- CLIP_GEN_PROCESS_GEN_CODE: backbone hidden state -> 16 RVQ codes + next-step embd ----664    // not build_inp_raw(), a GEN_WAV call's `img` has no hidden-state data665    ggml_tensor * h_state = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, n_mmproj_embd);666    ggml_set_name(h_state, "inp_raw"); // must keep this exact name, clip_encode() sets it by name667    ggml_set_input(h_state);668 669    ggml_tensor * code0 = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, 1);670    ggml_set_name(code0, "inp_code0");671    ggml_set_input(code0);672 673    ggml_tensor * code0_embd = ggml_get_rows(ctx0, model.gen_code_out_embd_w, code0);674    code0_embd = ggml_reshape_1d(ctx0, code0_embd, code0_embd->ne[0]);675    cb(code0_embd, "code0_embd", -1);676 677    const int64_t n_acoustic = model.gen_code_head_w->ne[2]; // 15678    const int     n_codes    = (int) n_acoustic + 1;         // 16679    const int64_t n_kv_pad   = n_codes;680    const int     n_layer    = (int) model.layers.size();681    const int     n_head     = hparams.n_head;682    const int     n_head_kv  = hparams.n_head_kv;683    const int64_t d_head     = model.layers[0].q_w->ne[1] / n_head;684 685    // zero-filled per layer k/v caches, so masked-out rows can't hold garbage686    std::vector<ggml_tensor *> k_cache(n_layer), v_cache(n_layer);687    for (int il = 0; il < n_layer; il++) {688        k_cache[il] = ggml_fill(ctx0, ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, d_head * n_head_kv, n_kv_pad), 0.0f);689        v_cache[il] = ggml_fill(ctx0, ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, d_head * n_head_kv, n_kv_pad), 0.0f);690    }691 692    code_gen cg(*this, top_k, top_p);693 694    ggml_tensor * out_code_cache = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, 1, n_codes);695    out_code_cache = cg.cache_set(out_code_cache, 0, code0);696 697    ggml_tensor * inp_rand0 = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, 1);698    ggml_set_name(inp_rand0, "inp_rand_0");699    ggml_set_input(inp_rand0);700 701    cg.prefill(k_cache, v_cache, out_code_cache, h_state, code0_embd, inp_rand0);702 703    for (int g = 1; g < n_acoustic; g++) {704        ggml_tensor * inp_rand = ggml_new_tensor_1d(ctx0, GGML_TYPE_F32, 1);705        ggml_set_name(inp_rand, ("inp_rand_" + std::to_string(g)).c_str());706        ggml_set_input(inp_rand);707        out_code_cache = cg.step(k_cache, v_cache, out_code_cache, inp_rand, g);708    }709 710    // output 1: this frame's 16 sampled codes, for the caller's code2wav window711    ggml_tensor * out_codes = ggml_cont(ctx0, out_code_cache);712    ggml_set_name(out_codes, "out_codes");713    ggml_set_output(out_codes);714 715    // output 2: sum of all 16 codebook embeddings, fed back to the talker for the next frame716    ggml_tensor * out_embd = code0_embd;717    for (int g = 1; g <= n_acoustic; g++) {718        ggml_tensor * code_g = ggml_view_1d(ctx0, out_code_cache, 1, (size_t) g * out_code_cache->nb[1]);719 720        ggml_tensor * embd_g = ggml_view_2d(ctx0, model.gen_code_embd_w, model.gen_code_embd_w->ne[0], model.gen_code_embd_w->ne[1],721                                            model.gen_code_embd_w->nb[1], (size_t) (g - 1) * model.gen_code_embd_w->nb[2]);722        ggml_tensor * e = ggml_get_rows(ctx0, embd_g, code_g);723        e = ggml_reshape_1d(ctx0, e, e->ne[0]);724 725        out_embd = ggml_add(ctx0, out_embd, e);726    }727    out_embd = ggml_reshape_2d(ctx0, out_embd, out_embd->ne[0], 1);728    cb(out_embd, "gen_audio_out", -1);729 730    // ---- CLIP_GEN_PROCESS_GEN_WAV: 16 RVQ codes -> raw PCM ----731    const int n_frames = hparams.wav_tfm_swa; // frames per batch, == the attention window732 733    ggml_tensor * inp_codes = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, n_frames, n_codes);734    ggml_set_name(inp_codes, "inp_codes");735    ggml_set_input(inp_codes);736 737    code2wav c2w(*this);738    for (const auto & slot : list_c2w_state_slots(hparams, model)) {739        ggml_tensor * t = ggml_new_tensor_2d(ctx0, GGML_TYPE_F32, slot.ne0, slot.ne1);740        ggml_set_name(t, ("state_in_" + slot.name).c_str());741        ggml_set_input(t);742        c2w.state_in[slot.name] = t;743    }744 745    ggml_tensor * out_audio = c2w.decode(inp_codes);746    ggml_set_name(out_audio, "out_audio");747    ggml_set_output(out_audio);748 749    for (auto & slot : c2w.state_out) {750        ggml_set_name(slot.second, ("state_out_" + slot.first).c_str());751        ggml_set_output(slot.second);752    }753 754    // out_embd goes last, clip_encode() reads it back via ggml_graph_node(gf, -1)755    ggml_tensor * outs[2];756    outs[0] = out_codes; outs[1] = out_audio;757    ggml_build_forward_select(gf, outs, 2, idx);758    for (auto & slot : c2w.state_out) {759        outs[0] = out_codes; outs[1] = slot.second;760        ggml_build_forward_select(gf, outs, 2, idx);761    }762    outs[0] = out_embd; outs[1] = out_audio;763    ggml_build_forward_select(gf, outs, 2, idx);764 765    return gf;766}767 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai