Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
granite4-vision.cpp340 linesDownload Raw Back to models
1#include "models.h"2#include "../clip-impl.h"3#include "../clip-model.h"4 5#include <algorithm>6#include <cmath>7#include <cstring>8#include <string>9#include <vector>10 11/*12 * Granite Vision 4.1 clip graph13 *14 *   Stage 1a: SigLIP vision tower (N layers, post-norm)15 *   Stage 1b: WindowQFormer blocks (deepstack + spatial)16 *   Stage 1c: Concatenate and pack outputs17 *   Stage 1d: Append newline tokens if add_newline is set18 */19 20// ---------------------------------------------------------------------------21// Member method implementations22// ---------------------------------------------------------------------------23 24ggml_tensor * clip_graph_granite4_vision::gather(25        ggml_tensor * src,26        const std::string & name,27        int idx_len) {28    ggml_tensor * idx = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, idx_len);29    ggml_set_name(idx, name.c_str());30    ggml_set_input(idx);31    return ggml_get_rows(ctx0, src, idx);32}33 34ggml_tensor * clip_graph_granite4_vision::interp_down(35        ggml_tensor * src,36        int side,37        int new_side) {38    const int n_embd = src->ne[0];39    ggml_tensor * t = ggml_reshape_4d(ctx0, src, n_embd, side, side, 1);40    t = ggml_cont(ctx0, ggml_permute(ctx0, t, 2, 0, 1, 3));41    const int kernel = side / new_side;42    t = ggml_pool_2d(ctx0, t, GGML_OP_POOL_AVG, kernel, kernel, kernel, kernel, 0, 0);43    t = ggml_cont(ctx0, ggml_permute(ctx0, t, 1, 2, 0, 3));44    return ggml_reshape_2d(ctx0, t, n_embd, new_side * new_side);45}46 47// ---------------------------------------------------------------------------48// build_block - WindowQFormer block implementation49// ---------------------------------------------------------------------------50 51ggml_tensor * clip_graph_granite4_vision::build_block(52        const qf_block & blk,53        ggml_tensor * h,54        int bid,55        int spatial_offset,56        int image_side,57        int window_side,58        int query_side,59        float qformer_eps) {60 61    const int n_embd = h->ne[0];62    GGML_ASSERT(h->ne[1] == image_side * image_side);63    const int n = image_side / window_side;64    const int new_side = n * query_side;65    const int n_windows = n * n;66    const int enc_len = window_side * window_side;67    const int query_len = query_side * query_side;68 69    auto cbx = [&](ggml_tensor * & t, const char * step) {70        const std::string name = "g4v_blk" + std::to_string(bid) + "_" + step;71        ggml_set_name(t, name.c_str());72    };73 74    // 1. Top-level LN75    cbx(h, "inp");76    ggml_tensor * x = build_norm(h, blk.qf_proj_norm_w, blk.qf_proj_norm_b, NORM_TYPE_NORMAL, eps, bid);77    cbx(x, "norm");78 79    // 2. enc = _win(x, image_side, window_side)80    ggml_tensor * enc;81    {82        ggml_tensor * enc_flat = gather(x,83            "g4v_blk" + std::to_string(bid) + "_win_idx",84            image_side * image_side);85        enc = ggml_reshape_3d(ctx0, enc_flat, n_embd, enc_len, n_windows);86    }87    cbx(enc, "enc");88 89    // 3. downsampled = downsampler(x)90    ggml_tensor * d;91    (void) spatial_offset;92    if (spatial_offset >= 0) {93        d = gather(x,94            "g4v_blk" + std::to_string(bid) + "_spatial_idx",95            new_side * new_side);96    } else {97        d = interp_down(x, image_side, new_side);98    }99    cbx(d, "downsampled");100 101    // 4. query_embeds = query + _win(d, new_side, query_side)102    ggml_tensor * q_in;103    {104        ggml_tensor * dw_flat = gather(d,105            "g4v_blk" + std::to_string(bid) + "_qwin_idx",106            new_side * new_side);107        ggml_tensor * dw = ggml_reshape_3d(ctx0, dw_flat, n_embd, query_len, n_windows);108        q_in = ggml_add(ctx0, dw, blk.qf_proj_query);109    }110    cbx(q_in, "query_embeds");111 112    // 5. encoder_embeds = enc + image_positions → (C, enc_len, n_windows)113    ggml_tensor * e_in = ggml_add(ctx0, enc, blk.qf_proj_img_pos);114    cbx(e_in, "encoder_embeds");115 116    // 6. Qformer forward.117    ggml_tensor * q = build_norm(q_in, blk.qf_proj_post_norm_w, blk.qf_proj_post_norm_b, NORM_TYPE_NORMAL, qformer_eps, bid);118 119    // Helper for linear projections with window batching120    auto linear = [&](ggml_tensor * x, ggml_tensor * w, ggml_tensor * b) -> ggml_tensor * {121        ggml_tensor * t = ggml_reshape_2d(ctx0, x, x->ne[0], x->ne[1] * x->ne[2]);122        t = build_mm(w, t);123        if (b) t = ggml_add(ctx0, t, b);124        return t;125    };126 127    // Get the single QFormer layer128    GGML_ASSERT(blk.qf_proj_layers.size() == 1);129    const auto & pl = blk.qf_proj_layers[0];130 131    // 6a. Self-attention132    ggml_tensor * sa_out;133    {134        const int d_h = 64;135        const int n_head = n_embd / d_h;136        const int nq = q->ne[1];137        const float scale = 1.0f / std::sqrt((float) d_h);138 139        ggml_tensor * Q = linear(q, pl.q_w, pl.q_b);140        ggml_tensor * K = linear(q, pl.k_w, pl.k_b);141        ggml_tensor * V = linear(q, pl.v_w, pl.v_b);142 143        Q = ggml_reshape_4d(ctx0, Q, d_h, n_head, nq, n_windows);144        K = ggml_reshape_4d(ctx0, K, d_h, n_head, nq, n_windows);145        V = ggml_reshape_4d(ctx0, V, d_h, n_head, nq, n_windows);146 147        sa_out = build_attn(pl.o_w, pl.o_b, Q, K, V, nullptr, scale, bid);148        sa_out = ggml_reshape_3d(ctx0, sa_out, n_embd, nq, n_windows);149 150        sa_out = ggml_add(ctx0, sa_out, q);151        sa_out = build_norm(sa_out, pl.ln_1_w, pl.ln_1_b,152                            NORM_TYPE_NORMAL, qformer_eps, bid);153    }154    cbx(sa_out, "sa_out");155 156    // 6b. Cross-attention157    ggml_tensor * ca_out;158    {159        const int d_h = 64;160        const int n_head = n_embd / d_h;161        const int nq = sa_out->ne[1];162        const int nkv = e_in->ne[1];163        const float scale = 1.0f / std::sqrt((float) d_h);164 165        ggml_tensor * Q = linear(sa_out, pl.cross_attn_q_w, pl.cross_attn_q_b);166        ggml_tensor * K = linear(e_in, pl.cross_attn_k_w, pl.cross_attn_k_b);167        ggml_tensor * V = linear(e_in, pl.cross_attn_v_w, pl.cross_attn_v_b);168 169        Q = ggml_reshape_4d(ctx0, Q, d_h, n_head, nq, n_windows);170        K = ggml_reshape_4d(ctx0, K, d_h, n_head, nkv, n_windows);171        V = ggml_reshape_4d(ctx0, V, d_h, n_head, nkv, n_windows);172 173        ca_out = build_attn(pl.cross_attn_o_w, pl.cross_attn_o_b,174                            Q, K, V, nullptr, scale, bid);175        ca_out = ggml_reshape_3d(ctx0, ca_out, n_embd, nq, n_windows);176 177        ca_out = ggml_add(ctx0, ca_out, sa_out);178        ca_out = build_norm(ca_out, pl.cross_attn_norm_w, pl.cross_attn_norm_b,179                            NORM_TYPE_NORMAL, qformer_eps, bid);180    }181    cbx(ca_out, "ca_out");182 183    // 6c. FFN184    ggml_tensor * ffn;185    {186        ggml_tensor * t = ggml_reshape_2d(ctx0, ca_out, n_embd, query_len * n_windows);187        t = build_mm(pl.ff_up_w, t);188        if (pl.ff_up_b) t = ggml_add(ctx0, t, pl.ff_up_b);189        t = ggml_gelu_erf(ctx0, t);190        t = build_mm(pl.ff_down_w, t);191        if (pl.ff_down_b) t = ggml_add(ctx0, t, pl.ff_down_b);192        t = ggml_reshape_3d(ctx0, t, n_embd, query_len, n_windows);193        ffn = ggml_add(ctx0, t, ca_out);194        ffn = build_norm(ffn, pl.ln_2_w, pl.ln_2_b, NORM_TYPE_NORMAL, qformer_eps, bid);195    }196    cbx(ffn, "qformer_out");197 198    // 7. _unwin back to raster199    ggml_tensor * unwinned;200    {201        ggml_tensor * flat = ggml_reshape_2d(ctx0, ffn, n_embd, query_len * n_windows);202        unwinned = gather(flat,203            "g4v_blk" + std::to_string(bid) + "_unwin_idx",204            new_side * new_side);205    }206    cbx(unwinned, "unwin");207 208    // 8. out_linear209    ggml_tensor * out = build_mm(blk.qf_proj_linear_w, unwinned);210    if (blk.qf_proj_linear_b) out = ggml_add(ctx0, out, blk.qf_proj_linear_b);211    cbx(out, "out");212 213    return out;214}215 216// ---------------------------------------------------------------------------217// build() - top-level graph218// ---------------------------------------------------------------------------219 220// Build the K-tiled, base-scaled newline row tensor.221// Shape: (n_mmproj_embd, 1)222ggml_tensor * clip_graph_granite4_vision::build_newline_row(ggml_context * ctx0) {223    const int K = (int) model.qf_proj_blocks.size();224    GGML_ASSERT(K > 0);225    GGML_ASSERT(n_mmproj_embd % K == 0);226    const int projection_dim = n_mmproj_embd / K;227    GGML_ASSERT(model.image_newline != nullptr);228    GGML_ASSERT(ggml_nelements(model.image_newline) == projection_dim);229 230    // Build newline_row[k*projection_dim + d] = nl[d] * (k == 0 ? base : 1.0)231    ggml_tensor * nl = model.image_newline; // (projection_dim,)232    ggml_tensor * nl_first_2d = ggml_reshape_2d(ctx0, nl, projection_dim, 1);233    ggml_tensor * nl_row_2d;234    if (K == 1) {235        nl_row_2d = nl_first_2d;236    } else {237        ggml_tensor * nl_2d = ggml_reshape_2d(ctx0, nl, projection_dim, 1);238        ggml_tensor * rest_template = ggml_new_tensor_2d(239            ctx0, GGML_TYPE_F32, projection_dim, K - 1);240        ggml_tensor * nl_rest = ggml_repeat(ctx0, nl_2d, rest_template);241        nl_row_2d = ggml_concat(ctx0, nl_first_2d, nl_rest, 1); // (projection_dim, K)242    }243    nl_row_2d = ggml_cont(ctx0, nl_row_2d);244    return ggml_reshape_2d(ctx0, nl_row_2d, n_mmproj_embd, 1);245}246 247// Append a single newline row at the end of the tile output.248ggml_tensor * clip_graph_granite4_vision::append_rowwise_newlines(ggml_context * ctx0, ggml_tensor * tile_output) {249    // For the single-tile case, append one newline row at the end.250    // For the multi-tile rowwise case, this will be called per-tile251    // (though currently only the single-tile path uses it).252    ggml_tensor * nl_row = build_newline_row(ctx0);253    return ggml_concat(ctx0, tile_output, nl_row, 1);254}255 256ggml_cgraph * clip_graph_granite4_vision::build() {257    GGML_ASSERT(model.patch_embeddings_0 != nullptr);258    GGML_ASSERT(model.position_embeddings != nullptr);259    GGML_ASSERT(model.class_embedding == nullptr);260    GGML_ASSERT(!model.qf_proj_blocks.empty());261 262    // --- Stage 1a: SigLIP encoder producing intermediate hidden states ---263    ggml_tensor * inp = build_inp();264    inp = ggml_add(ctx0, inp, model.position_embeddings);265    cb(inp, "pos_embed", -1);266 267    ggml_tensor * inpL = inp;268    std::vector<ggml_tensor *> layer_outs(n_layer, nullptr);269 270    for (int il = 0; il < n_layer; ++il) {271        const auto & layer = model.layers[il];272        ggml_tensor * cur = inpL;273 274        cur = build_norm(cur, layer.ln_1_w, layer.ln_1_b, NORM_TYPE_NORMAL, eps, il);275 276        // Self-attention277        ggml_tensor * Qcur = build_mm(layer.q_w, cur);278        if (layer.q_b) Qcur = ggml_add(ctx0, Qcur, layer.q_b);279        ggml_tensor * Kcur = build_mm(layer.k_w, cur);280        if (layer.k_b) Kcur = ggml_add(ctx0, Kcur, layer.k_b);281        ggml_tensor * Vcur = build_mm(layer.v_w, cur);282        if (layer.v_b) Vcur = ggml_add(ctx0, Vcur, layer.v_b);283 284        Qcur = ggml_reshape_3d(ctx0, Qcur, d_head, n_head, n_patches);285        Kcur = ggml_reshape_3d(ctx0, Kcur, d_head, n_head, n_patches);286        Vcur = ggml_reshape_3d(ctx0, Vcur, d_head, n_head, n_patches);287 288        cur = build_attn(layer.o_w, layer.o_b,289                         Qcur, Kcur, Vcur, nullptr, kq_scale, il);290 291        cur = ggml_add(ctx0, cur, inpL);292        inpL = cur;293 294        cur = build_norm(cur, layer.ln_2_w, layer.ln_2_b, NORM_TYPE_NORMAL, eps, il);295        cur = build_ffn(cur,296                        layer.ff_up_w, layer.ff_up_b,297                        layer.ff_gate_w, layer.ff_gate_b,298                        layer.ff_down_w, layer.ff_down_b,299                        hparams.ffn_op, il);300        cur = ggml_add(ctx0, inpL, cur);301        cb(cur, "layer_out", il);302        layer_outs[il] = cur;303        inpL = cur;304    }305 306    // --- Stage 1b/1c: WindowQFormer blocks ---307    const int projector_count = hparams.feature_layers.size();308    const float qformer_eps = 1e-12f;309 310    ggml_tensor * mmproj = nullptr;311    for (int bid = 0; bid < projector_count; ++bid) {312        const auto & blk = model.qf_proj_blocks[bid];313 314        int vlayer = hparams.feature_layers[bid];315        GGML_ASSERT(vlayer >= 0 && vlayer < n_layer);316        ggml_tensor * h = layer_outs[vlayer];317 318        ggml_tensor * stream = build_block(319            blk, h, bid,320            hparams.proj_spatial_offsets[bid],321            n_patches_x,322            hparams.downsample_window_side,323            hparams.downsample_query_side,324            qformer_eps);325        cb(stream, (std::string("proj_") + std::to_string(bid) + std::string("_v_out")).c_str(), vlayer);326        mmproj = mmproj ? ggml_concat(ctx0, mmproj, stream, 0) : stream;327    }328 329    // --- Stage 1d: Append newline tokens if add_newline is set ---330    if (add_newline) {331        mmproj = append_rowwise_newlines(ctx0, mmproj);332        ggml_set_name(mmproj, "g4v_mmproj_out_nl");333    } else {334        ggml_set_name(mmproj, "g4v_mmproj_out");335    }336    ggml_build_forward_expand(gf, mmproj);337 338    return gf;339}340 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai