Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
llava.cpp374 linesDownload Raw Back to models
1#include "models.h"2 3// this graph is used by llava, granite and glm4// due to having embedding_stack (used by granite), we cannot reuse build_vit5ggml_cgraph * clip_graph_llava::build() {6    const int batch_size = 1;7    const int n_pos = n_patches + (model.class_embedding ? 1 : 0);8 9    GGML_ASSERT(n_patches_x == n_patches_y && "only square images supported");10 11    // Calculate the deepest feature layer based on hparams and projector type12    int max_feature_layer = n_layer;13    {14        // Get the index of the second to last layer; this is the default for models that have a llava projector15        int il_last = hparams.n_layer - 1;16        int deepest_feature_layer = -1;17 18        if (proj_type == PROJECTOR_TYPE_MINICPMV || proj_type == PROJECTOR_TYPE_GLM_EDGE) {19            il_last += 1;20        }21 22        // If we set explicit vision feature layers, only go up to the deepest one23        // NOTE: only used by granite-vision models for now24        for (const auto & feature_layer : hparams.feature_layers) {25            if (feature_layer > deepest_feature_layer) {26                deepest_feature_layer = feature_layer;27            }28        }29        max_feature_layer = deepest_feature_layer < 0 ? il_last : deepest_feature_layer;30    }31 32    ggml_tensor * inp = build_inp();33 34    // concat class_embeddings and patch_embeddings35    if (model.class_embedding) {36        inp = ggml_concat(ctx0, inp, model.class_embedding, 1);37    }38 39    ggml_tensor * positions = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_pos);40    ggml_set_name(positions, "positions");41    ggml_set_input(positions);42 43    inp = ggml_add(ctx0, inp, ggml_get_rows(ctx0, model.position_embeddings, positions));44 45    ggml_tensor * inpL = inp;46 47    // pre-layernorm48    if (model.pre_ln_w) {49        inpL = build_norm(inpL, model.pre_ln_w, model.pre_ln_b, NORM_TYPE_NORMAL, eps, -1);50        cb(inpL, "pre_ln", -1);51    }52 53    std::vector<ggml_tensor *> embedding_stack;54 55    // loop over layers56    for (int il = 0; il < max_feature_layer; il++) {57        auto & layer = model.layers[il];58        ggml_tensor * cur = inpL; // inpL = residual, cur = hidden_states59 60        // If this is an embedding feature layer, save the output.61        // NOTE: 0 index here refers to the input to the encoder.62        if (hparams.is_feature_layer(il)) {63            embedding_stack.push_back(cur);64        }65 66        // layernorm167        cur = build_norm(cur, layer.ln_1_w, layer.ln_1_b, NORM_TYPE_NORMAL, eps, il);68        cb(cur, "layer_inp_normed", il);69 70        // self-attention71        {72            ggml_tensor * Qcur = build_mm(layer.q_w, cur);73            if (layer.q_b) {74                Qcur = ggml_add(ctx0, Qcur, layer.q_b);75            }76 77            ggml_tensor * Kcur = build_mm(layer.k_w, cur);78            if (layer.k_b) {79                Kcur = ggml_add(ctx0, Kcur, layer.k_b);80            }81 82            ggml_tensor * Vcur = build_mm(layer.v_w, cur);83            if (layer.v_b) {84                Vcur = ggml_add(ctx0, Vcur, layer.v_b);85            }86 87            Qcur = ggml_reshape_3d(ctx0, Qcur, d_head, n_head, n_pos);88            Kcur = ggml_reshape_3d(ctx0, Kcur, d_head, n_head, n_pos);89            Vcur = ggml_reshape_3d(ctx0, Vcur, d_head, n_head, n_pos);90 91            cb(Qcur, "Qcur", il);92            cb(Kcur, "Kcur", il);93            cb(Vcur, "Vcur", il);94 95            cur = build_attn(layer.o_w, layer.o_b,96                Qcur, Kcur, Vcur, nullptr, kq_scale, il);97            cb(cur, "attn_out", il);98        }99 100        // re-add the layer input, e.g., residual101        cur = ggml_add(ctx0, cur, inpL);102 103        inpL = cur; // inpL = residual, cur = hidden_states104 105        cb(cur, "ffn_inp", il);106 107        // layernorm2108        cur = build_norm(cur, layer.ln_2_w, layer.ln_2_b, NORM_TYPE_NORMAL, eps, il);109        cb(cur, "ffn_inp_normed", il);110 111        // ffn112        cur = build_ffn(cur,113            layer.ff_up_w, layer.ff_up_b,114            layer.ff_gate_w, layer.ff_gate_b,115            layer.ff_down_w, layer.ff_down_b,116            hparams.ffn_op, il);117 118        cb(cur, "ffn_out", il);119 120        // residual 2121        cur = ggml_add(ctx0, inpL, cur);122        cb(cur, "layer_out", il);123 124        inpL = cur;125    }126 127    // post-layernorm128    if (model.post_ln_w) {129        inpL = build_norm(inpL, model.post_ln_w, model.post_ln_b, NORM_TYPE_NORMAL, eps, -1);130    }131 132    ggml_tensor * embeddings = inpL;133 134    // process vision feature layers (used by granite)135    {136        // final layer is a vision feature layer137        if (hparams.is_feature_layer(max_feature_layer)) {138            embedding_stack.push_back(inpL);139        }140 141        // If feature layers are explicitly set, stack them (if we have multiple)142        if (!embedding_stack.empty()) {143            embeddings = embedding_stack[0];144            for (size_t i = 1; i < embedding_stack.size(); i++) {145                embeddings = ggml_concat(ctx0, embeddings, embedding_stack[i], 0);146            }147        }148    }149 150    // llava projector (also used by granite)151    if (hparams.has_llava_projector) {152        embeddings = ggml_reshape_2d(ctx0, embeddings, embeddings->ne[0], embeddings->ne[1]);153 154        ggml_tensor * patches = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, n_patches);155        ggml_set_name(patches, "patches");156        ggml_set_input(patches);157 158        // shape [1, 576, 1024]159        // ne is whcn, ne = [1024, 576, 1, 1]160        embeddings = ggml_get_rows(ctx0, embeddings, patches);161 162        // print_tensor_info(embeddings, "embeddings");163 164        // llava projector165        if (proj_type == PROJECTOR_TYPE_MLP) {166            embeddings = build_mm(model.mm_0_w, embeddings);167            embeddings = ggml_add(ctx0, embeddings, model.mm_0_b);168 169            embeddings = ggml_gelu(ctx0, embeddings);170            if (model.mm_2_w) {171                embeddings = build_mm(model.mm_2_w, embeddings);172                embeddings = ggml_add(ctx0, embeddings, model.mm_2_b);173            }174        }175        else if (proj_type == PROJECTOR_TYPE_MLP_NORM) {176            embeddings = build_mm(model.mm_0_w, embeddings);177            embeddings = ggml_add(ctx0, embeddings, model.mm_0_b);178            // ggml_tensor_printf(embeddings, "mm_0_w",0,true,false);179            // First LayerNorm180            embeddings = ggml_norm(ctx0, embeddings, eps);181            embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_1_w),182                                model.mm_1_b);183 184            // GELU activation185            embeddings = ggml_gelu(ctx0, embeddings);186 187            // Second linear layer188            embeddings = build_mm(model.mm_3_w, embeddings);189            embeddings = ggml_add(ctx0, embeddings, model.mm_3_b);190 191            // Second LayerNorm192            embeddings = ggml_norm(ctx0, embeddings, eps);193            embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_4_w),194                                model.mm_4_b);195        }196        else if (proj_type == PROJECTOR_TYPE_LDP) {197            // MobileVLM projector198            int n_patch = 24;199            ggml_tensor * mlp_1 = build_mm(model.mm_model_mlp_1_w, embeddings);200            mlp_1 = ggml_add(ctx0, mlp_1, model.mm_model_mlp_1_b);201            mlp_1 = ggml_gelu(ctx0, mlp_1);202            ggml_tensor * mlp_3 = build_mm(model.mm_model_mlp_3_w, mlp_1);203            mlp_3 = ggml_add(ctx0, mlp_3, model.mm_model_mlp_3_b);204            // mlp_3 shape = [1, 576, 2048], ne = [2048, 576, 1, 1]205 206            // block 1207            ggml_tensor * block_1 = nullptr;208            {209                // transpose from [1, 576, 2048] --> [1, 2048, 576] --> [1, 2048, 24, 24]210                mlp_3 = ggml_permute(ctx0, mlp_3, 1, 0, 2, 3);211                mlp_3 = ggml_cont_4d(ctx0, mlp_3, n_patch, n_patch, mlp_3->ne[1], mlp_3->ne[2]);212                // stride = 1, padding = 1, bias is nullptr213                block_1 = ggml_conv_2d_dw(ctx0, model.mm_model_block_1_block_0_0_w, mlp_3, 1, 1, 1, 1, 1, 1);214 215                // layer norm216                // // block_1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]217                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 2, 0, 3));218                // block_1 shape = [1, 24, 24, 2048], ne = [2048, 24, 24, 1]219                block_1 = ggml_norm(ctx0, block_1, eps);220                block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_1_block_0_1_w), model.mm_model_block_1_block_0_1_b);221                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));222 223                // block_1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]224                // hardswish225                ggml_tensor * block_1_hw = ggml_hardswish(ctx0, block_1);226 227                block_1 = ggml_pool_2d(ctx0, block_1_hw, GGML_OP_POOL_AVG, block_1_hw->ne[0], block_1_hw->ne[1], block_1_hw->ne[0], block_1_hw->ne[1], 0, 0);228                // block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]229                // pointwise conv230                block_1 = ggml_reshape_2d(ctx0, block_1, block_1->ne[0]*block_1->ne[1]*block_1->ne[2], block_1->ne[3]);231                block_1 = build_mm(model.mm_model_block_1_block_1_fc1_w, block_1);232                block_1 = ggml_add(ctx0, block_1, model.mm_model_block_1_block_1_fc1_b);233                block_1 = ggml_relu(ctx0, block_1);234                block_1 = build_mm(model.mm_model_block_1_block_1_fc2_w, block_1);235                block_1 = ggml_add(ctx0, block_1, model.mm_model_block_1_block_1_fc2_b);236                block_1 = ggml_hardsigmoid(ctx0, block_1);237                // block_1_hw shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1], block_1 shape = [1, 2048], ne = [2048, 1, 1, 1]238                block_1 = ggml_reshape_4d(ctx0, block_1, 1, 1, block_1->ne[0], block_1->ne[1]);239                block_1 = ggml_mul(ctx0, block_1_hw, block_1);240 241                int w = block_1->ne[0], h = block_1->ne[1];242                block_1 = ggml_reshape_3d(ctx0, block_1, w*h, block_1->ne[2], block_1->ne[3]);243                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 0, 2, 3));244 245                // block_1 shape = [1, 24*24, 2048], ne = [24*24, 2048, 1]246                block_1 = build_mm(model.mm_model_block_1_block_2_0_w, block_1);247                block_1 = ggml_reshape_4d(ctx0, block_1, block_1->ne[0], w, h, block_1->ne[3]);248 249                // block_1 shape = [1, 24, 24, 2048], ne = [2048, 24, 24, 1]250                block_1 = ggml_norm(ctx0, block_1, eps);251                block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_1_block_2_1_w), model.mm_model_block_1_block_2_1_b);252                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));253                // block1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]254                // residual255                block_1 = ggml_add(ctx0, mlp_3, block_1);256            }257 258            // block_2259            {260                // stride = 2261                block_1 = ggml_conv_2d_dw(ctx0, model.mm_model_block_2_block_0_0_w, block_1, 2, 2, 1, 1, 1, 1);262 263                // block_1 shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1]264                // layer norm265                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 2, 0, 3));266                // block_1 shape = [1, 12, 12, 2048], ne = [2048, 12, 12, 1]267                block_1 = ggml_norm(ctx0, block_1, eps);268                block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_2_block_0_1_w), model.mm_model_block_2_block_0_1_b);269                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));270                // block_1 shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1]271                // hardswish272                ggml_tensor * block_1_hw = ggml_hardswish(ctx0, block_1);273 274                // not sure the parameters is right for globalAvgPooling275                block_1 = ggml_pool_2d(ctx0, block_1_hw, GGML_OP_POOL_AVG, block_1_hw->ne[0], block_1_hw->ne[1], block_1_hw->ne[0], block_1_hw->ne[1], 0, 0);276                // block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]277                // pointwise conv278                block_1 = ggml_reshape_2d(ctx0, block_1, block_1->ne[0]*block_1->ne[1]*block_1->ne[2], block_1->ne[3]);279                block_1 = build_mm(model.mm_model_block_2_block_1_fc1_w, block_1);280                block_1 = ggml_add(ctx0, block_1, model.mm_model_block_2_block_1_fc1_b);281                block_1 = ggml_relu(ctx0, block_1);282                block_1 = build_mm(model.mm_model_block_2_block_1_fc2_w, block_1);283                block_1 = ggml_add(ctx0, block_1, model.mm_model_block_2_block_1_fc2_b);284                block_1 = ggml_hardsigmoid(ctx0, block_1);285 286                // block_1_hw shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1], block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]287                block_1 = ggml_reshape_4d(ctx0, block_1, 1, 1, block_1->ne[0], block_1->ne[1]);288                block_1 = ggml_mul(ctx0, block_1_hw, block_1);289 290                int w = block_1->ne[0], h = block_1->ne[1];291                block_1 = ggml_reshape_3d(ctx0, block_1, w*h, block_1->ne[2], block_1->ne[3]);292                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 0, 2, 3));293                // block_1 shape = [1, 24*24, 2048], ne = [24*24, 2048, 1]294                block_1 = build_mm(model.mm_model_block_2_block_2_0_w, block_1);295                block_1 = ggml_reshape_4d(ctx0, block_1, block_1->ne[0], w, h, block_1->ne[3]);296 297 298                // block_1 shape = [1, 12, 12, 2048], ne = [2048, 12, 12, 1]299                block_1 = ggml_norm(ctx0, block_1, eps);300                block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_2_block_2_1_w), model.mm_model_block_2_block_2_1_b);301                block_1 = ggml_reshape_3d(ctx0, block_1, block_1->ne[0], block_1->ne[1] * block_1->ne[2], block_1->ne[3]);302                // block_1 shape = [1, 144, 2048], ne = [2048, 144, 1]303            }304            embeddings = block_1;305        }306        else if (proj_type == PROJECTOR_TYPE_LDPV2)307        {308            int n_patch = 24;309            ggml_tensor * mlp_0 = build_mm(model.mm_model_mlp_0_w, embeddings);310            mlp_0 = ggml_add(ctx0, mlp_0, model.mm_model_mlp_0_b);311            mlp_0 = ggml_gelu(ctx0, mlp_0);312            ggml_tensor * mlp_2 = build_mm(model.mm_model_mlp_2_w, mlp_0);313            mlp_2 = ggml_add(ctx0, mlp_2, model.mm_model_mlp_2_b);314            // mlp_2 ne = [2048, 576, 1, 1]315            // // AVG Pool Layer 2*2, strides = 2316            mlp_2 = ggml_permute(ctx0, mlp_2, 1, 0, 2, 3);317            // mlp_2 ne = [576, 2048, 1, 1]318            mlp_2 = ggml_cont_4d(ctx0, mlp_2, n_patch, n_patch, mlp_2->ne[1], mlp_2->ne[2]);319            // mlp_2 ne [24, 24, 2048, 1]320            mlp_2 = ggml_pool_2d(ctx0, mlp_2, GGML_OP_POOL_AVG, 2, 2, 2, 2, 0, 0);321            // weight ne = [3, 3, 2048, 1]322            ggml_tensor * peg_0 = ggml_conv_2d_dw(ctx0, model.mm_model_peg_0_w, mlp_2, 1, 1, 1, 1, 1, 1);323            peg_0 = ggml_cont(ctx0, ggml_permute(ctx0, peg_0, 1, 2, 0, 3));324            peg_0 = ggml_add(ctx0, peg_0, model.mm_model_peg_0_b);325            mlp_2 = ggml_cont(ctx0, ggml_permute(ctx0, mlp_2, 1, 2, 0, 3));326            peg_0 = ggml_add(ctx0, peg_0, mlp_2);327            peg_0 = ggml_reshape_3d(ctx0, peg_0, peg_0->ne[0], peg_0->ne[1] * peg_0->ne[2], peg_0->ne[3]);328            embeddings = peg_0;329        }330        else {331            GGML_ABORT("fatal error");332        }333    }334 335    // glm projector336    else if (proj_type == PROJECTOR_TYPE_GLM_EDGE) {337        size_t gridsz = (size_t)sqrt(embeddings->ne[1]);338        embeddings = ggml_permute(ctx0,embeddings,1,0,2,3);339        embeddings = ggml_cont_3d(ctx0, embeddings, gridsz, gridsz, embeddings->ne[1]);340        embeddings = ggml_conv_2d(ctx0, model.mm_model_adapter_conv_w, embeddings, 2, 2, 0, 0, 1, 1);341        embeddings = ggml_reshape_3d(ctx0, embeddings,embeddings->ne[0]*embeddings->ne[1] , embeddings->ne[2], batch_size);342        embeddings = ggml_cont(ctx0, ggml_permute(ctx0,embeddings, 1, 0, 2, 3));343        embeddings = ggml_add(ctx0, embeddings, model.mm_model_adapter_conv_b);344        // GLU345        {346            embeddings = build_mm(model.mm_model_mlp_0_w, embeddings);347            embeddings = ggml_norm(ctx0, embeddings, eps);348            embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_model_ln_q_w), model.mm_model_ln_q_b);349            embeddings = ggml_gelu_inplace(ctx0, embeddings);350            ggml_tensor * x = embeddings;351            embeddings = build_mm(model.mm_model_mlp_2_w, embeddings);352            x = build_mm(model.mm_model_mlp_1_w,x);353            embeddings = ggml_swiglu_split(ctx0, embeddings, x);354            embeddings = build_mm(model.mm_model_mlp_3_w, embeddings);355        }356        // arrangement of BOI/EOI token embeddings357        // note: these embeddings are not present in text model, hence we cannot process them as text tokens358        // see: https://huggingface.co/THUDM/glm-edge-v-2b/blob/main/siglip.py#L53359        {360            embeddings = ggml_concat(ctx0, model.mm_boi, embeddings, 1); // BOI361            embeddings = ggml_concat(ctx0, embeddings, model.mm_eoi, 1); // EOI362        }363    }364 365    else {366        GGML_ABORT("llava: unknown projector type");367    }368 369    // build the graph370    ggml_build_forward_expand(gf, embeddings);371 372    return gf;373}374 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai