Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
deepseekocr.cpp399 linesDownload Raw Back to models
1#include "models.h"2 3// Implementation based on approach suggested by Acly4// See: https://github.com/ggml-org/llama.cpp/pull/17383#issuecomment-35542270915static ggml_tensor * window_partition(ggml_context * ctx0, ggml_tensor * x, const int window) {6    auto [c, w, h, b] = x->ne;7    // same as8    // x = ggml_win_part(m, x, window);9    // x = ggml_reshape_3d(m, x, c, window * window, x->ne[3]);10 11    const int64_t px  = (window - w % window) % window;12    const int64_t py  = (window - h % window) % window;13    const int64_t npw = (w + px) / window;14    const int64_t nph = (h + py) / window;15 16    ggml_tensor * cur = x;17    if (px > 0 || py > 0) {18        cur = ggml_pad(ctx0, cur, 0, static_cast<int>(px), static_cast<int>(py), 0);19    }20    cur = ggml_reshape_4d(ctx0, cur, c * window, npw, window, nph * b);21    cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 0, 2, 1, 3));22    cur = ggml_reshape_4d(ctx0, cur, c, window, window, npw * nph * b);23    return cur;24}25 26// Implementation based on approach suggested by Acly27// See: https://github.com/ggml-org/llama.cpp/pull/17383#issuecomment-355422709128static ggml_tensor * window_unpartition(ggml_context * ctx0,29                                        ggml_tensor *  x,30                                        const int      w,31                                        const int      h,32                                        const int      window) {33    const int64_t c = x->ne[0];34    // same as35    // x = ggml_reshape_4d(m, x, c, window, window, x->ne[2]);36    // x = ggml_win_unpart(m, x, w, h, window);37 38    const int64_t px  = (window - w % window) % window;39    const int64_t py  = (window - h % window) % window;40    const int64_t npw = (w + px) / window;41    const int64_t nph = (h + py) / window;42 43    const int64_t b = x->ne[3] / (npw * nph);44    ggml_tensor * cur = x;45    cur = ggml_reshape_4d(ctx0, cur, c * window, window, npw, nph * b);46    cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 0, 2, 1, 3));47    cur = ggml_reshape_4d(ctx0, cur, c, w + px, h + py, b);48    cur = ggml_view_4d(ctx0, cur, cur->ne[0], w, h, cur->ne[3], cur->nb[1], cur->nb[2], cur->nb[3], 0);49    cur = ggml_cont(ctx0, cur);50    return cur;51}52 53static ggml_tensor * get_rel_pos(ggml_context * ctx0,54                                 ggml_tensor *  rel_pos,  // [L, C]55                                 ggml_tensor *  indices,  // [q_size, k_size]56                                 const int      q_size,57                                 const int      k_size) {58    const int64_t C = rel_pos->ne[0];  // channels59    const int64_t L = rel_pos->ne[1];  // length60 61    GGML_ASSERT(indices != nullptr);62    GGML_ASSERT(indices->type == GGML_TYPE_I32);63    GGML_ASSERT(indices->ne[0] == k_size);64    GGML_ASSERT(indices->ne[1] == q_size);65 66    const auto    max_rel_dist = 2 * std::max(q_size, k_size) - 1;67    ggml_tensor * cur          = rel_pos;68 69    if (max_rel_dist != L) {70        // Linear interpolation71        const int64_t ne0 = cur->ne[0];72        const int64_t ne1 = cur->ne[1];73        const int64_t ne2 = cur->ne[2];74        const int64_t ne3 = cur->ne[3];75 76        cur = ggml_reshape_3d(ctx0, ggml_cont(ctx0, ggml_permute(ctx0, cur, 1, 0, 2, 3)), ne1, 1, ne0 * ne2 * ne3);77        cur = ggml_reshape_4d(78            ctx0, ggml_interpolate(ctx0, cur, max_rel_dist, 1, ne0 * ne2 * ne3, 1, GGML_SCALE_MODE_BILINEAR),79            max_rel_dist, ne0, ne2, ne3);80        cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 1, 0, 2, 3));81    }82 83    // Flatten indices to 1D for ggml_get_rows84    const int qk = q_size * k_size;85 86    cur = ggml_reshape_3d(ctx0, ggml_get_rows(ctx0, cur, ggml_reshape_1d(ctx0, indices, qk)), C, k_size, q_size);87 88    return cur;  // [C, k_size, q_size]89}90 91 92ggml_tensor * clip_graph_deepseekocr::build_sam(ggml_tensor * inp_raw) {93    // Building SAM94    const int n_embd  = hparams.sam_n_embd;95    const int n_layer = hparams.sam_n_layer;96    const int n_heads = hparams.sam_n_head;97    const int d_heads = n_embd / n_heads;98    const int window  = hparams.attn_window_size;99    // SAM stage runs its layernorms at 1e-6100    const float sam_eps = 1e-6f;101 102    ggml_tensor * inpL;103 104    inpL = ggml_conv_2d_sk_p0(ctx0, model.patch_embed_proj_w, inp_raw);105    inpL = ggml_add(ctx0, inpL, ggml_reshape_3d(ctx0, model.patch_embed_proj_b, 1, 1, n_embd));106    inpL = ggml_cont(ctx0, ggml_permute(ctx0, inpL, 1, 2, 0, 3));107 108    ggml_tensor * rel_pos_indices_local;109    ggml_tensor * rel_pos_indices_global;110 111    rel_pos_indices_local  = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, window, window);112    rel_pos_indices_global = ggml_new_tensor_2d(ctx0, GGML_TYPE_I32, inpL->ne[1], inpL->ne[2]);113    ggml_set_name(rel_pos_indices_local, "rel_pos_indices_local");114    ggml_set_name(rel_pos_indices_global, "rel_pos_indices_global");115    ggml_set_input(rel_pos_indices_local);116    ggml_set_input(rel_pos_indices_global);117 118    ggml_tensor * cur;119    const auto    tgt_size = inpL->ne[1];120    const auto    str_size = model.pos_embed->ne[1];121 122    if (str_size != tgt_size) {123        ggml_tensor * old_pos_embed = nullptr;124        old_pos_embed               = ggml_cont(ctx0, ggml_permute(ctx0, model.pos_embed, 2, 0, 1, 3));125        ggml_tensor * new_pos_embed =126            ggml_interpolate(ctx0, old_pos_embed, tgt_size, tgt_size, n_embd, 1, GGML_SCALE_MODE_BICUBIC);127        new_pos_embed = ggml_cont(ctx0, ggml_permute(ctx0, new_pos_embed, 1, 2, 0, 3));128        cur           = ggml_add(ctx0, inpL, new_pos_embed);129    } else {130        cur = ggml_add(ctx0, inpL, model.pos_embed);131    }132 133    // loop over layers134    for (int il = 0; il < n_layer; il++) {135        auto &        layer    = model.sam_layers[il];136        ggml_tensor * shortcut = cur;137 138        // layernorm1139        cur = build_norm(cur, layer.ln_1_w, layer.ln_1_b, NORM_TYPE_NORMAL, sam_eps, il);140 141        const int64_t w0 = cur->ne[1];142        const int64_t h0 = cur->ne[2];143 144        ggml_tensor * indices;145 146        if (hparams.is_global_attn(il)) {147            indices = rel_pos_indices_global;148        } else {149            // local attention layer - apply window partition150            cur     = window_partition(ctx0, cur, window);151            indices = rel_pos_indices_local;152        }153 154        const int64_t W = cur->ne[1];155        const int64_t H = cur->ne[2];156        // self-attention157        {158            const int B = cur->ne[3];159 160            cur = ggml_mul_mat(ctx0, layer.qkv_w, cur);161            cur = ggml_add(ctx0, cur, layer.qkv_b);162            cur = ggml_reshape_4d(ctx0, cur, n_embd, 3, W * H, B);163 164            ggml_tensor * Q;165            ggml_tensor * K;166            ggml_tensor * V;167 168            Q = ggml_view_3d(ctx0, cur, n_embd, W * H, B, cur->nb[2], cur->nb[3], 0 * cur->nb[1]);169            Q = ggml_reshape_4d(ctx0, ggml_cont(ctx0, Q), d_heads, n_heads, W * H, B);170 171            K = ggml_view_3d(ctx0, cur, n_embd, W * H, B, cur->nb[2], cur->nb[3], 1 * cur->nb[1]);172            K = ggml_reshape_4d(ctx0, ggml_cont(ctx0, K), d_heads, n_heads, W * H, B);173 174            V = ggml_view_3d(ctx0, cur, n_embd, W * H, B, cur->nb[2], cur->nb[3], 2 * cur->nb[1]);175            V = ggml_reshape_4d(ctx0, ggml_cont(ctx0, V), d_heads, n_heads, W * H, B);176 177            ggml_tensor * mask;178            ggml_tensor * rw;179            ggml_tensor * rh;180            ggml_tensor * qr;181 182            rw = get_rel_pos(ctx0, layer.rel_pos_w, indices, W, W); // [W, W, C]183            rh = get_rel_pos(ctx0, layer.rel_pos_h, indices, H, H); // [H, H, C]184            qr = ggml_permute(ctx0, Q, 0, 2, 1, 3);185            qr = ggml_reshape_4d(ctx0, ggml_cont(ctx0, qr), d_heads, W, H, B * n_heads);186 187            rw = ggml_mul_mat(ctx0, rw,188                              ggml_cont(ctx0, ggml_permute(ctx0, qr, 0, 2, 1, 3))); // [B*n_heads, W, H, W]189            rw   = ggml_cont(ctx0, ggml_permute(ctx0, rw, 0, 2, 1, 3)); // [B*n_heads, H, W, W]190            rw   = ggml_reshape_4d(ctx0, rw, W, 1, W * H, n_heads * B);191            rw   = ggml_repeat_4d(ctx0, rw, W, H, W * H, n_heads * B);192            rh   = ggml_mul_mat(ctx0, rh, qr); // [B*n_heads, H, W, H]193            rh   = ggml_reshape_4d(ctx0, rh, 1, H, W * H, n_heads * B);194            mask = ggml_add(ctx0, rw, rh); // [B*n_heads, H*W, H, W]195            mask = ggml_reshape_4d(ctx0, mask, W * H, W * H, n_heads, B);196            // casting mask to F16 only required when flash-attn is enabled197            if (flash_attn_type == CLIP_FLASH_ATTN_TYPE_ENABLED) {198                mask = ggml_cast(ctx0, mask, GGML_TYPE_F16);199            }200 201            const float scale = 1.0f / sqrtf(static_cast<float>(d_heads));202 203            cur = build_attn(layer.o_w, layer.o_b, Q, K, V, mask, scale,204                             il); // [B, H*W, n_embd]205            cur = ggml_reshape_4d(ctx0, ggml_cont(ctx0, cur), n_embd, W, H, B);206        }207 208        if (hparams.is_global_attn(il) == false) {209            // local attention layer - reverse window partition210            cur = window_unpartition(ctx0, cur, w0, h0, window);211        }212 213        // re-add the layer input, e.g., residual214        cur = ggml_add(ctx0, cur, shortcut);215 216        ggml_tensor * inpFF = cur;217 218        // layernorm2219        cur = build_norm(inpFF, layer.ln_2_w, layer.ln_2_b, NORM_TYPE_NORMAL, sam_eps, il);220 221        // ffn222        cur = build_ffn(cur, layer.ff_up_w, layer.ff_up_b, nullptr, nullptr, layer.ff_down_w, layer.ff_down_b,223                        hparams.ffn_op, il);224 225        // residual 2226        cur = ggml_add(ctx0, cur, inpFF);227        cb(cur, "sam_layer_out", il);228    }229 230    cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 2, 0, 1, 3));231 232    cur = ggml_conv_2d(ctx0, model.neck_0_w, cur, 1, 1, 0, 0, 1, 1);233    cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 1, 2, 0, 3));234    cur = build_norm(cur, model.neck_1_w, model.neck_1_b, NORM_TYPE_NORMAL, sam_eps, -1);235    cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 2, 0, 1, 3));236 237    cur = ggml_conv_2d(ctx0, model.neck_2_w, cur, 1, 1, 1, 1, 1, 1);238    cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 1, 2, 0, 3));239    cur = build_norm(cur, model.neck_3_w, model.neck_3_b, NORM_TYPE_NORMAL, sam_eps, -1);240    cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 2, 0, 1, 3));241 242    cur = ggml_conv_2d(ctx0, model.net_2, cur, 2, 2, 1, 1, 1, 1);243    cur = ggml_conv_2d(ctx0, model.net_3, cur, 2, 2, 1, 1, 1, 1);244    cb(cur, "sam_output", -1);245 246    ggml_build_forward_expand(gf, cur);247    return cur;248}249 250ggml_cgraph * clip_graph_deepseekocr::build() {251    // patch embedding252    ggml_tensor * inp_raw = build_inp_raw();253 254    bool is_overview = img.add_viewsep;255    int n_tiles_per_row = 0;256    // number of separate "row" images batched together in this graph call257    // (captured now, before n_batch below gets repurposed as the SAM/ViT batch size)258    const int n_rows_batch = n_batch;259 260    // note: we expect either a batch of rows or a batch of overviews, but not a mix of both261 262    if (!is_overview) {263        // handle the case where we have a batch of rows264        // sanity check265        for (auto & entry : img_batch->entries) {266            if (entry.add_viewsep) {267                throw std::runtime_error("DeepSeek-OCR: mixed overview and non-overview images in batch");268            }269            if (entry.nx() != img.nx() || entry.ny() != img.ny()) {270                throw std::runtime_error("DeepSeek-OCR: mixed image sizes in batch");271            }272        }273 274        GGML_ASSERT(img.ny() >= img.nx());275        GGML_ASSERT(img.ny() % img.nx() == 0);276        n_tiles_per_row = img.ny() / img.nx();277 278        // each entry is one "row" image of shape [tile_size, tile_size * n_tiles_per_row, 3];279        // merge the tile axis into the batch axis, giving a combined SAM input of shape280        // [tile_size, tile_size, 3, n_tiles_per_row * n_rows_batch] (tile fast, row slow)281        inp_raw = ggml_reshape_4d(ctx0, inp_raw, img.nx() * img.nx(), n_tiles_per_row, 3, n_rows_batch);282        inp_raw = ggml_cont(ctx0, ggml_permute(ctx0, inp_raw, 0, 2, 1, 3));283        inp_raw = ggml_reshape_4d(ctx0, inp_raw, img.nx(), img.nx(), 3, n_tiles_per_row * n_rows_batch);284    }285 286    ggml_tensor * sam_out = build_sam(inp_raw);287 288    if (!is_overview) {289        n_batch = n_tiles_per_row * n_rows_batch;290    }291 292    const int clip_n_patches = sam_out->ne[0] * sam_out->ne[1];293 294    ggml_tensor * clip_out;295    // Building DS-OCR CLIP296    {297        ggml_tensor * inp;298 299        // sam_out: [patch_h, patch_w, n_embd, n_batch]300        // -> [n_embd, clip_n_patches, n_batch]301        inp = ggml_reshape_3d(ctx0, sam_out, clip_n_patches, sam_out->ne[2], sam_out->ne[3]);302        inp = ggml_cont(ctx0, ggml_permute(ctx0, inp, 1, 0, 2, 3));303 304        ggml_tensor * new_pos_embd = model.position_embeddings;305 306        int        n_pos    = new_pos_embd->ne[1];  // +1 for [CLS]307        const auto tgt_size = static_cast<int>(std::sqrt(inp->ne[1]));308        const auto src_size = static_cast<int>(std::sqrt(n_pos - 1));309 310        if (tgt_size != src_size) {311            ggml_tensor * old_pos_embd;312            ggml_tensor * cls_tok;313 314            old_pos_embd = ggml_view_2d(ctx0, new_pos_embd, new_pos_embd->ne[0], src_size * src_size,315                                        ggml_row_size(new_pos_embd->type, new_pos_embd->ne[0]), 0);316            cls_tok      = ggml_view_2d(ctx0, new_pos_embd, new_pos_embd->ne[0], 1,317                                        ggml_row_size(new_pos_embd->type, new_pos_embd->ne[0]), src_size * src_size);318            new_pos_embd = ggml_interpolate(ctx0, old_pos_embd, tgt_size, tgt_size, new_pos_embd->ne[0], 1,319                                            GGML_SCALE_MODE_BICUBIC);320            new_pos_embd = ggml_reshape_3d(ctx0, new_pos_embd, n_embd, tgt_size * tgt_size, 1);321            new_pos_embd = ggml_concat(ctx0, new_pos_embd, cls_tok, 1);322            n_pos        = tgt_size * tgt_size + 1;323        }324 325        // add CLS token per batch item326        // inp: [n_embd, clip_n_patches, n_batch]327        // class_embedding: [n_embd] -> [n_embd, 1, n_batch]328        ggml_tensor * cls_embd = ggml_repeat_4d(ctx0, model.class_embedding, n_embd, 1, n_batch, 1);329        inp = ggml_concat(ctx0, cls_embd, inp, 1);330 331        // for selecting learned pos embd, used by ViT332        ggml_tensor * positions        = ggml_cast(ctx0, ggml_arange(ctx0, 0, n_pos, 1), GGML_TYPE_I32);333        ggml_tensor * learned_pos_embd = ggml_get_rows(ctx0, new_pos_embd, positions);334 335        ggml_tensor * cur = build_vit(inp, n_pos, NORM_TYPE_NORMAL, FFN_GELU_QUICK, learned_pos_embd, nullptr);336 337        ggml_build_forward_expand(gf, cur);338        clip_out = cur;339    }340 341    // sam_out: [patch_h, patch_w, n_embd, n_batch]342    // -> [n_embd, clip_n_patches, n_batch]343    sam_out  = ggml_cont(ctx0, ggml_permute(ctx0, sam_out, 1, 2, 0, 3));344    sam_out  = ggml_reshape_3d(ctx0, sam_out, sam_out->ne[0], clip_n_patches, n_batch);345 346    // clip_out: [n_embd, n_pos, n_batch] where n_pos = clip_n_patches + 1 (CLS)347    // strip CLS token: skip first position, view only the patch tokens348    clip_out = ggml_view_3d(ctx0, clip_out, n_embd, clip_n_patches, n_batch,349                            clip_out->nb[1], clip_out->nb[2], clip_out->nb[1]);350 351    ggml_tensor * cur;352    cur = ggml_concat(ctx0, clip_out, sam_out, 0);353    cur = ggml_mul_mat(ctx0, model.mm_fc_w, cur);354    cur = ggml_add(ctx0, cur, model.mm_fc_b);355 356    if (is_overview) {357        // global view: weave one newline per row + trailing view separator358        const auto h     = static_cast<int>(std::sqrt(static_cast<float>(cur->ne[1])));359        const auto w     = h;360        const auto n_dim = cur->ne[0];361 362        ggml_tensor * imgnl = ggml_repeat_4d(ctx0, model.image_newline, n_dim, 1, h, n_batch);363        cur = ggml_reshape_4d(ctx0, cur, n_dim, w, h, n_batch);364        cur = ggml_reshape_3d(ctx0, ggml_concat(ctx0, cur, imgnl, 1), n_dim, (w + 1) * h, n_batch);365        ggml_tensor * vs = ggml_repeat_4d(ctx0, model.view_seperator, n_dim, 1, n_batch, 1);366        cur = ggml_concat(ctx0, cur, vs, 1);  // (n_dim, h*(w+1) + 1, n_batch)367    } else {368        // tile row: interleave tiles within each row, add newline per row369        const int  grid_x = static_cast<int>(std::sqrt(static_cast<float>(clip_n_patches)));370        const int  grid_y = grid_x;371        const auto n_dim  = cur->ne[0];372 373        // merge n_dim into the grid_x axis, freeing the 4th axis for n_rows_batch374        // (n_dim, clip_n_patches, n_tiles_per_row * n_rows_batch) -> (n_dim*grid_x, grid_y, n_tiles_per_row, n_rows_batch)375        cur = ggml_reshape_4d(ctx0, cur, n_dim * grid_x, grid_y, n_tiles_per_row, n_rows_batch);376 377        // tiles: re-order from A.row0 A.row1 B.row0 B.row1 ...378        //        to A.row0 B.row0 A.row1 B.row1 ...379        //        then add nl: A.row0 B.row0 [nl] A.row1 B.row1 [nl] ...380        // interleave tiles: -> (n_dim*grid_x, n_tiles_per_row, grid_y, n_rows_batch)381        cur = ggml_cont(ctx0, ggml_permute(ctx0, cur, 0, 2, 1, 3));382 383        // merge: -> (n_dim, grid_x*n_tiles_per_row, grid_y, n_rows_batch)384        cur = ggml_reshape_4d(ctx0, cur, n_dim, grid_x * n_tiles_per_row, grid_y, n_rows_batch);385 386        // append newline per row: (n_dim, grid_x*n_tiles_per_row+1, grid_y, n_rows_batch)387        ggml_tensor * imgnl = ggml_repeat_4d(ctx0, model.image_newline, n_dim, 1, grid_y, n_rows_batch);388        cur = ggml_concat(ctx0, cur, imgnl, 1);389 390        // flatten: (n_dim, (grid_x*n_tiles_per_row+1)*grid_y, n_rows_batch)391        cur = ggml_reshape_3d(ctx0, cur, n_dim, (grid_x * n_tiles_per_row + 1) * grid_y, n_rows_batch);392    }393 394    cb(cur, "dsocr_output", -1);395 396    ggml_build_forward_expand(gf, cur);397    return gf;398}399 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai