Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#include "models.h"2#include "../clip-impl.h"3#include "../clip-model.h"4 5#include <algorithm>6#include <cmath>7#include <cstring>8#include <string>9#include <vector>10 11/*12 * Granite Vision 4.1 clip graph13 *14 * Stage 1a: SigLIP vision tower (N layers, post-norm)15 * Stage 1b: WindowQFormer blocks (deepstack + spatial)16 * Stage 1c: Concatenate and pack outputs17 * Stage 1d: Append newline tokens if add_newline is set18 */19 20// ---------------------------------------------------------------------------21// Member method implementations22// ---------------------------------------------------------------------------23 24ggml_tensor * clip_graph_granite4_vision::gather(25 ggml_tensor * src,26 const std::string & name,27 int idx_len) {28 ggml_tensor * idx = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, idx_len);29 ggml_set_name(idx, name.c_str());30 ggml_set_input(idx);31 return ggml_get_rows(ctx0, src, idx);32}33 34ggml_tensor * clip_graph_granite4_vision::interp_down(35 ggml_tensor * src,36 int side,37 int new_side) {38 const int n_embd = src->ne[0];39 ggml_tensor * t = ggml_reshape_4d(ctx0, src, n_embd, side, side, 1);40 t = ggml_cont(ctx0, ggml_permute(ctx0, t, 2, 0, 1, 3));41 const int kernel = side / new_side;42 t = ggml_pool_2d(ctx0, t, GGML_OP_POOL_AVG, kernel, kernel, kernel, kernel, 0, 0);43 t = ggml_cont(ctx0, ggml_permute(ctx0, t, 1, 2, 0, 3));44 return ggml_reshape_2d(ctx0, t, n_embd, new_side * new_side);45}46 47// ---------------------------------------------------------------------------48// build_block - WindowQFormer block implementation49// ---------------------------------------------------------------------------50 51ggml_tensor * clip_graph_granite4_vision::build_block(52 const qf_block & blk,53 ggml_tensor * h,54 int bid,55 int spatial_offset,56 int image_side,57 int window_side,58 int query_side,59 float qformer_eps) {60 61 const int n_embd = h->ne[0];62 GGML_ASSERT(h->ne[1] == image_side * image_side);63 const int n = image_side / window_side;64 const int new_side = n * query_side;65 const int n_windows = n * n;66 const int enc_len = window_side * window_side;67 const int query_len = query_side * query_side;68 69 auto cbx = [&](ggml_tensor * & t, const char * step) {70 const std::string name = "g4v_blk" + std::to_string(bid) + "_" + step;71 ggml_set_name(t, name.c_str());72 };73 74 // 1. Top-level LN75 cbx(h, "inp");76 ggml_tensor * x = build_norm(h, blk.qf_proj_norm_w, blk.qf_proj_norm_b, NORM_TYPE_NORMAL, eps, bid);77 cbx(x, "norm");78 79 // 2. enc = _win(x, image_side, window_side)80 ggml_tensor * enc;81 {82 ggml_tensor * enc_flat = gather(x,83 "g4v_blk" + std::to_string(bid) + "_win_idx",84 image_side * image_side);85 enc = ggml_reshape_3d(ctx0, enc_flat, n_embd, enc_len, n_windows);86 }87 cbx(enc, "enc");88 89 // 3. downsampled = downsampler(x)90 ggml_tensor * d;91 (void) spatial_offset;92 if (spatial_offset >= 0) {93 d = gather(x,94 "g4v_blk" + std::to_string(bid) + "_spatial_idx",95 new_side * new_side);96 } else {97 d = interp_down(x, image_side, new_side);98 }99 cbx(d, "downsampled");100 101 // 4. query_embeds = query + _win(d, new_side, query_side)102 ggml_tensor * q_in;103 {104 ggml_tensor * dw_flat = gather(d,105 "g4v_blk" + std::to_string(bid) + "_qwin_idx",106 new_side * new_side);107 ggml_tensor * dw = ggml_reshape_3d(ctx0, dw_flat, n_embd, query_len, n_windows);108 q_in = ggml_add(ctx0, dw, blk.qf_proj_query);109 }110 cbx(q_in, "query_embeds");111 112 // 5. encoder_embeds = enc + image_positions → (C, enc_len, n_windows)113 ggml_tensor * e_in = ggml_add(ctx0, enc, blk.qf_proj_img_pos);114 cbx(e_in, "encoder_embeds");115 116 // 6. Qformer forward.117 ggml_tensor * q = build_norm(q_in, blk.qf_proj_post_norm_w, blk.qf_proj_post_norm_b, NORM_TYPE_NORMAL, qformer_eps, bid);118 119 // Helper for linear projections with window batching120 auto linear = [&](ggml_tensor * x, ggml_tensor * w, ggml_tensor * b) -> ggml_tensor * {121 ggml_tensor * t = ggml_reshape_2d(ctx0, x, x->ne[0], x->ne[1] * x->ne[2]);122 t = build_mm(w, t);123 if (b) t = ggml_add(ctx0, t, b);124 return t;125 };126 127 // Get the single QFormer layer128 GGML_ASSERT(blk.qf_proj_layers.size() == 1);129 const auto & pl = blk.qf_proj_layers[0];130 131 // 6a. Self-attention132 ggml_tensor * sa_out;133 {134 const int d_h = 64;135 const int n_head = n_embd / d_h;136 const int nq = q->ne[1];137 const float scale = 1.0f / std::sqrt((float) d_h);138 139 ggml_tensor * Q = linear(q, pl.q_w, pl.q_b);140 ggml_tensor * K = linear(q, pl.k_w, pl.k_b);141 ggml_tensor * V = linear(q, pl.v_w, pl.v_b);142 143 Q = ggml_reshape_4d(ctx0, Q, d_h, n_head, nq, n_windows);144 K = ggml_reshape_4d(ctx0, K, d_h, n_head, nq, n_windows);145 V = ggml_reshape_4d(ctx0, V, d_h, n_head, nq, n_windows);146 147 sa_out = build_attn(pl.o_w, pl.o_b, Q, K, V, nullptr, scale, bid);148 sa_out = ggml_reshape_3d(ctx0, sa_out, n_embd, nq, n_windows);149 150 sa_out = ggml_add(ctx0, sa_out, q);151 sa_out = build_norm(sa_out, pl.ln_1_w, pl.ln_1_b,152 NORM_TYPE_NORMAL, qformer_eps, bid);153 }154 cbx(sa_out, "sa_out");155 156 // 6b. Cross-attention157 ggml_tensor * ca_out;158 {159 const int d_h = 64;160 const int n_head = n_embd / d_h;161 const int nq = sa_out->ne[1];162 const int nkv = e_in->ne[1];163 const float scale = 1.0f / std::sqrt((float) d_h);164 165 ggml_tensor * Q = linear(sa_out, pl.cross_attn_q_w, pl.cross_attn_q_b);166 ggml_tensor * K = linear(e_in, pl.cross_attn_k_w, pl.cross_attn_k_b);167 ggml_tensor * V = linear(e_in, pl.cross_attn_v_w, pl.cross_attn_v_b);168 169 Q = ggml_reshape_4d(ctx0, Q, d_h, n_head, nq, n_windows);170 K = ggml_reshape_4d(ctx0, K, d_h, n_head, nkv, n_windows);171 V = ggml_reshape_4d(ctx0, V, d_h, n_head, nkv, n_windows);172 173 ca_out = build_attn(pl.cross_attn_o_w, pl.cross_attn_o_b,174 Q, K, V, nullptr, scale, bid);175 ca_out = ggml_reshape_3d(ctx0, ca_out, n_embd, nq, n_windows);176 177 ca_out = ggml_add(ctx0, ca_out, sa_out);178 ca_out = build_norm(ca_out, pl.cross_attn_norm_w, pl.cross_attn_norm_b,179 NORM_TYPE_NORMAL, qformer_eps, bid);180 }181 cbx(ca_out, "ca_out");182 183 // 6c. FFN184 ggml_tensor * ffn;185 {186 ggml_tensor * t = ggml_reshape_2d(ctx0, ca_out, n_embd, query_len * n_windows);187 t = build_mm(pl.ff_up_w, t);188 if (pl.ff_up_b) t = ggml_add(ctx0, t, pl.ff_up_b);189 t = ggml_gelu_erf(ctx0, t);190 t = build_mm(pl.ff_down_w, t);191 if (pl.ff_down_b) t = ggml_add(ctx0, t, pl.ff_down_b);192 t = ggml_reshape_3d(ctx0, t, n_embd, query_len, n_windows);193 ffn = ggml_add(ctx0, t, ca_out);194 ffn = build_norm(ffn, pl.ln_2_w, pl.ln_2_b, NORM_TYPE_NORMAL, qformer_eps, bid);195 }196 cbx(ffn, "qformer_out");197 198 // 7. _unwin back to raster199 ggml_tensor * unwinned;200 {201 ggml_tensor * flat = ggml_reshape_2d(ctx0, ffn, n_embd, query_len * n_windows);202 unwinned = gather(flat,203 "g4v_blk" + std::to_string(bid) + "_unwin_idx",204 new_side * new_side);205 }206 cbx(unwinned, "unwin");207 208 // 8. out_linear209 ggml_tensor * out = build_mm(blk.qf_proj_linear_w, unwinned);210 if (blk.qf_proj_linear_b) out = ggml_add(ctx0, out, blk.qf_proj_linear_b);211 cbx(out, "out");212 213 return out;214}215 216// ---------------------------------------------------------------------------217// build() - top-level graph218// ---------------------------------------------------------------------------219 220// Build the K-tiled, base-scaled newline row tensor.221// Shape: (n_mmproj_embd, 1)222ggml_tensor * clip_graph_granite4_vision::build_newline_row(ggml_context * ctx0) {223 const int K = (int) model.qf_proj_blocks.size();224 GGML_ASSERT(K > 0);225 GGML_ASSERT(n_mmproj_embd % K == 0);226 const int projection_dim = n_mmproj_embd / K;227 GGML_ASSERT(model.image_newline != nullptr);228 GGML_ASSERT(ggml_nelements(model.image_newline) == projection_dim);229 230 // Build newline_row[k*projection_dim + d] = nl[d] * (k == 0 ? base : 1.0)231 ggml_tensor * nl = model.image_newline; // (projection_dim,)232 ggml_tensor * nl_first_2d = ggml_reshape_2d(ctx0, nl, projection_dim, 1);233 ggml_tensor * nl_row_2d;234 if (K == 1) {235 nl_row_2d = nl_first_2d;236 } else {237 ggml_tensor * nl_2d = ggml_reshape_2d(ctx0, nl, projection_dim, 1);238 ggml_tensor * rest_template = ggml_new_tensor_2d(239 ctx0, GGML_TYPE_F32, projection_dim, K - 1);240 ggml_tensor * nl_rest = ggml_repeat(ctx0, nl_2d, rest_template);241 nl_row_2d = ggml_concat(ctx0, nl_first_2d, nl_rest, 1); // (projection_dim, K)242 }243 nl_row_2d = ggml_cont(ctx0, nl_row_2d);244 return ggml_reshape_2d(ctx0, nl_row_2d, n_mmproj_embd, 1);245}246 247// Append a single newline row at the end of the tile output.248ggml_tensor * clip_graph_granite4_vision::append_rowwise_newlines(ggml_context * ctx0, ggml_tensor * tile_output) {249 // For the single-tile case, append one newline row at the end.250 // For the multi-tile rowwise case, this will be called per-tile251 // (though currently only the single-tile path uses it).252 ggml_tensor * nl_row = build_newline_row(ctx0);253 return ggml_concat(ctx0, tile_output, nl_row, 1);254}255 256ggml_cgraph * clip_graph_granite4_vision::build() {257 GGML_ASSERT(model.patch_embeddings_0 != nullptr);258 GGML_ASSERT(model.position_embeddings != nullptr);259 GGML_ASSERT(model.class_embedding == nullptr);260 GGML_ASSERT(!model.qf_proj_blocks.empty());261 262 // --- Stage 1a: SigLIP encoder producing intermediate hidden states ---263 ggml_tensor * inp = build_inp();264 inp = ggml_add(ctx0, inp, model.position_embeddings);265 cb(inp, "pos_embed", -1);266 267 ggml_tensor * inpL = inp;268 std::vector<ggml_tensor *> layer_outs(n_layer, nullptr);269 270 for (int il = 0; il < n_layer; ++il) {271 const auto & layer = model.layers[il];272 ggml_tensor * cur = inpL;273 274 cur = build_norm(cur, layer.ln_1_w, layer.ln_1_b, NORM_TYPE_NORMAL, eps, il);275 276 // Self-attention277 ggml_tensor * Qcur = build_mm(layer.q_w, cur);278 if (layer.q_b) Qcur = ggml_add(ctx0, Qcur, layer.q_b);279 ggml_tensor * Kcur = build_mm(layer.k_w, cur);280 if (layer.k_b) Kcur = ggml_add(ctx0, Kcur, layer.k_b);281 ggml_tensor * Vcur = build_mm(layer.v_w, cur);282 if (layer.v_b) Vcur = ggml_add(ctx0, Vcur, layer.v_b);283 284 Qcur = ggml_reshape_3d(ctx0, Qcur, d_head, n_head, n_patches);285 Kcur = ggml_reshape_3d(ctx0, Kcur, d_head, n_head, n_patches);286 Vcur = ggml_reshape_3d(ctx0, Vcur, d_head, n_head, n_patches);287 288 cur = build_attn(layer.o_w, layer.o_b,289 Qcur, Kcur, Vcur, nullptr, kq_scale, il);290 291 cur = ggml_add(ctx0, cur, inpL);292 inpL = cur;293 294 cur = build_norm(cur, layer.ln_2_w, layer.ln_2_b, NORM_TYPE_NORMAL, eps, il);295 cur = build_ffn(cur,296 layer.ff_up_w, layer.ff_up_b,297 layer.ff_gate_w, layer.ff_gate_b,298 layer.ff_down_w, layer.ff_down_b,299 hparams.ffn_op, il);300 cur = ggml_add(ctx0, inpL, cur);301 cb(cur, "layer_out", il);302 layer_outs[il] = cur;303 inpL = cur;304 }305 306 // --- Stage 1b/1c: WindowQFormer blocks ---307 const int projector_count = hparams.feature_layers.size();308 const float qformer_eps = 1e-12f;309 310 ggml_tensor * mmproj = nullptr;311 for (int bid = 0; bid < projector_count; ++bid) {312 const auto & blk = model.qf_proj_blocks[bid];313 314 int vlayer = hparams.feature_layers[bid];315 GGML_ASSERT(vlayer >= 0 && vlayer < n_layer);316 ggml_tensor * h = layer_outs[vlayer];317 318 ggml_tensor * stream = build_block(319 blk, h, bid,320 hparams.proj_spatial_offsets[bid],321 n_patches_x,322 hparams.downsample_window_side,323 hparams.downsample_query_side,324 qformer_eps);325 cb(stream, (std::string("proj_") + std::to_string(bid) + std::string("_v_out")).c_str(), vlayer);326 mmproj = mmproj ? ggml_concat(ctx0, mmproj, stream, 0) : stream;327 }328 329 // --- Stage 1d: Append newline tokens if add_newline is set ---330 if (add_newline) {331 mmproj = append_rowwise_newlines(ctx0, mmproj);332 ggml_set_name(mmproj, "g4v_mmproj_out_nl");333 } else {334 ggml_set_name(mmproj, "g4v_mmproj_out");335 }336 ggml_build_forward_expand(gf, mmproj);337 338 return gf;339}340 