Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
clip.cpp2941 linesDownload Raw Back to llava
1// NOTE: This is modified from clip.cpp only for LLaVA,2// so there might be still unnecessary artifacts hanging around3// I'll gradually clean and extend it4// Note: Even when using identical normalized image inputs (see normalize_image_u8_to_f32()) we have a significant difference in resulting embeddings compared to pytorch5#include "clip.h"6#include "ggml.h"7#include "ggml-cpu.h"8#include "ggml-alloc.h"9#include "ggml-backend.h"10#include "gguf.h"11 12//#ifdef GGML_USE_CUDA13//#include "ggml-cuda.h"14//#endif15//16//#ifdef GGML_USE_SYCL17//#include "ggml-sycl.h"18//#endif19//20//#ifdef GGML_USE_METAL21//#include "ggml-metal.h"22//#endif23//24//#ifdef GGML_USE_CANN25//#include "ggml-cann.h"26//#endif27//28//#ifdef GGML_USE_VULKAN29//#include "ggml-vulkan.h"30//#endif31 32#define STB_IMAGE_IMPLEMENTATION33#include "stb_image.h"34 35#include <cassert>36#include <cmath>37#include <cstdlib>38#include <cstring>39#include <fstream>40#include <map>41#include <regex>42#include <stdexcept>43#include <vector>44#include <sstream>45#include <cinttypes>46#include <limits>47 48#if defined(LLAVA_LOG_OFF)49#   define LOG_INF(...)50#   define LOG_WRN(...)51#   define LOG_ERR(...)52#   define LOG_DBG(...)53#else // defined(LLAVA_LOG_OFF)54#   define LOG_INF(...) do { fprintf(stdout, __VA_ARGS__); } while (0)55#   define LOG_WRN(...) do { fprintf(stderr, __VA_ARGS__); } while (0)56#   define LOG_ERR(...) do { fprintf(stderr, __VA_ARGS__); } while (0)57#   define LOG_DBG(...) do { fprintf(stdout, __VA_ARGS__); } while (0)58#endif // defined(LLAVA_LOG_OFF)59 60//#define CLIP_DEBUG_FUNCTIONS61 62// RGB uint8 image63struct clip_image_u8 {64    int nx;65    int ny;66 67    std::vector<uint8_t> buf;68};69 70// RGB float32 image (NHWC)71// Memory layout: RGBRGBRGB...72struct clip_image_f32 {73    int nx;74    int ny;75 76    std::vector<float> buf;77};78 79static std::string format(const char * fmt, ...) {80    va_list ap;81    va_list ap2;82    va_start(ap, fmt);83    va_copy(ap2, ap);84    int size = vsnprintf(NULL, 0, fmt, ap);85    GGML_ASSERT(size >= 0 && size < INT_MAX); // NOLINT86    std::vector<char> buf(size + 1);87    int size2 = vsnprintf(buf.data(), size + 1, fmt, ap2);88    GGML_ASSERT(size2 == size);89    va_end(ap2);90    va_end(ap);91    return std::string(buf.data(), buf.size());92}93 94//95// key constants96//97 98#define KEY_FTYPE               "general.file_type"99#define KEY_NAME                "general.name"100#define KEY_DESCRIPTION         "general.description"101#define KEY_HAS_TEXT_ENC        "clip.has_text_encoder"102#define KEY_HAS_VIS_ENC         "clip.has_vision_encoder"103#define KEY_HAS_LLAVA_PROJ      "clip.has_llava_projector"104#define KEY_HAS_MINICPMV_PROJ   "clip.has_minicpmv_projector"105#define KEY_HAS_GLM_PROJ        "clip.has_glm_projector"106#define KEY_MINICPMV_VERSION    "clip.minicpmv_version"107#define KEY_HAS_QWEN2VL_MERGER  "clip.has_qwen2vl_merger"108#define KEY_USE_GELU            "clip.use_gelu"109#define KEY_USE_SILU            "clip.use_silu"110#define KEY_N_EMBD              "clip.%s.embedding_length"111#define KEY_N_FF                "clip.%s.feed_forward_length"112#define KEY_N_BLOCK             "clip.%s.block_count"113#define KEY_N_HEAD              "clip.%s.attention.head_count"114#define KEY_LAYER_NORM_EPS      "clip.%s.attention.layer_norm_epsilon"115#define KEY_PROJ_DIM            "clip.%s.projection_dim"116#define KEY_TOKENS              "tokenizer.ggml.tokens"117#define KEY_N_POSITIONS         "clip.text.context_length"118#define KEY_IMAGE_SIZE          "clip.vision.image_size"119#define KEY_PATCH_SIZE          "clip.vision.patch_size"120#define KEY_IMAGE_MEAN          "clip.vision.image_mean"121#define KEY_IMAGE_STD           "clip.vision.image_std"122#define KEY_PROJ_TYPE           "clip.projector_type"123 124#define KEY_MM_PATCH_MERGE_TYPE   "clip.vision.mm_patch_merge_type"125#define KEY_IMAGE_GRID_PINPOINTS  "clip.vision.image_grid_pinpoints"126#define KEY_IMAGE_CROP_RESOLUTION "clip.vision.image_crop_resolution"127 128 129//130// tensor name constants131//132 133#define TN_TOKEN_EMBD      "%s.token_embd.weight"134#define TN_POS_EMBD        "%s.position_embd.weight"135#define TN_CLASS_EMBD      "v.class_embd"136#define TN_PATCH_EMBD      "v.patch_embd.weight"  // not rename tensor with ".0" postfix for backwrad compat137#define TN_PATCH_EMBD_1    "v.patch_embd.weight.1"138#define TN_PATCH_BIAS      "v.patch_embd.bias"139#define TN_ATTN_K          "%s.blk.%d.attn_k.%s"140#define TN_ATTN_Q          "%s.blk.%d.attn_q.%s"141#define TN_ATTN_V          "%s.blk.%d.attn_v.%s"142#define TN_ATTN_OUTPUT     "%s.blk.%d.attn_out.%s"143#define TN_FFN_DOWN        "%s.blk.%d.ffn_down.%s"144#define TN_FFN_UP          "%s.blk.%d.ffn_up.%s"145#define TN_LN_1            "%s.blk.%d.ln1.%s"146#define TN_LN_2            "%s.blk.%d.ln2.%s"147#define TN_LN_PRE          "%s.pre_ln.%s"148#define TN_LN_POST         "%s.post_ln.%s"149#define TN_TEXT_PROJ       "text_projection.weight"150#define TN_VIS_PROJ        "visual_projection.weight"151#define TN_LLAVA_PROJ      "mm.%d.%s"152#define TN_MVLM_PROJ_MLP   "mm.model.mlp.%d.%s"153#define TN_MVLM_PROJ_BLOCK "mm.model.mb_block.%d.block.%d.%s"154#define TN_MVLM_PROJ_PEG   "mm.model.peg.%d.%s"155#define TN_IMAGE_NEWLINE   "model.image_newline"156 157#define TN_MINICPMV_POS_EMBD_K "resampler.pos_embed_k"158#define TN_MINICPMV_QUERY "resampler.query"159#define TN_MINICPMV_PROJ "resampler.proj.weight"160#define TN_MINICPMV_KV_PROJ "resampler.kv.weight"161#define TN_MINICPMV_ATTN "resampler.attn.%s.%s"162#define TN_MINICPMV_LN "resampler.ln_%s.%s"163 164#define TN_GLM_ADAPER_CONV "adapter.conv.%s"165#define TN_GLM_ADAPTER_LINEAR "adapter.linear.linear.%s"166#define TN_GLM_ADAPTER_NORM_1 "adapter.linear.norm1.%s"167#define TN_GLM_ADAPTER_D_H_2_4H "adapter.linear.dense_h_to_4h.%s"168#define TN_GLM_ADAPTER_GATE "adapter.linear.gate.%s"169#define TN_GLM_ADAPTER_D_4H_2_H "adapter.linear.dense_4h_to_h.%s"170#define TN_GLM_BOI_W "adapter.boi"171#define TN_GLM_EOI_W "adapter.eoi"172 173 174enum projector_type {175    PROJECTOR_TYPE_MLP,176    PROJECTOR_TYPE_MLP_NORM,177    PROJECTOR_TYPE_LDP,178    PROJECTOR_TYPE_LDPV2,179    PROJECTOR_TYPE_RESAMPLER,180    PROJECTOR_TYPE_GLM_EDGE,181    PROJECTOR_TYPE_MERGER,182    PROJECTOR_TYPE_UNKNOWN,183};184 185static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {186    { PROJECTOR_TYPE_MLP, "mlp" },187    { PROJECTOR_TYPE_LDP, "ldp" },188    { PROJECTOR_TYPE_LDPV2, "ldpv2"},189    { PROJECTOR_TYPE_RESAMPLER, "resampler"},190    { PROJECTOR_TYPE_GLM_EDGE, "adapter"},191    { PROJECTOR_TYPE_MERGER, "qwen2vl_merger"},192};193 194 195//196// utilities to get data from a gguf file197//198 199static int get_key_idx(const gguf_context * ctx, const char * key) {200    int i = gguf_find_key(ctx, key);201    if (i == -1) {202        LOG_ERR("key %s not found in file\n", key);203        throw std::runtime_error(format("Missing required key: %s", key));204    }205 206    return i;207}208 209static uint32_t get_u32(const gguf_context * ctx, const std::string & key) {210    const int i = get_key_idx(ctx, key.c_str());211 212    return gguf_get_val_u32(ctx, i);213}214 215static float get_f32(const gguf_context * ctx, const std::string & key) {216    const int i = get_key_idx(ctx, key.c_str());217 218    return gguf_get_val_f32(ctx, i);219}220 221static struct ggml_tensor * get_tensor(struct ggml_context * ctx, const std::string & name) {222    struct ggml_tensor * cur = ggml_get_tensor(ctx, name.c_str());223    if (!cur) {224        throw std::runtime_error(format("%s: unable to find tensor %s\n", __func__, name.c_str()));225    }226 227    return cur;228}229 230static std::string get_ftype(int ftype) {231    return ggml_type_name(static_cast<ggml_type>(ftype));232}233 234static std::string gguf_data_to_str(enum gguf_type type, const void * data, int i) {235    switch (type) {236        case GGUF_TYPE_UINT8:   return std::to_string(((const uint8_t  *)data)[i]);237        case GGUF_TYPE_INT8:    return std::to_string(((const int8_t   *)data)[i]);238        case GGUF_TYPE_UINT16:  return std::to_string(((const uint16_t *)data)[i]);239        case GGUF_TYPE_INT16:   return std::to_string(((const int16_t  *)data)[i]);240        case GGUF_TYPE_UINT32:  return std::to_string(((const uint32_t *)data)[i]);241        case GGUF_TYPE_INT32:   return std::to_string(((const int32_t  *)data)[i]);242        case GGUF_TYPE_UINT64:  return std::to_string(((const uint64_t *)data)[i]);243        case GGUF_TYPE_INT64:   return std::to_string(((const int64_t  *)data)[i]);244        case GGUF_TYPE_FLOAT32: return std::to_string(((const float    *)data)[i]);245        case GGUF_TYPE_FLOAT64: return std::to_string(((const double   *)data)[i]);246        case GGUF_TYPE_BOOL:    return ((const bool *)data)[i] ? "true" : "false";247        default:                return format("unknown type %d", type);248    }249}250 251static void replace_all(std::string & s, const std::string & search, const std::string & replace) {252    if (search.empty()) {253        return;254    }255    std::string builder;256    builder.reserve(s.length());257    size_t pos = 0;258    size_t last_pos = 0;259    while ((pos = s.find(search, last_pos)) != std::string::npos) {260        builder.append(s, last_pos, pos - last_pos);261        builder.append(replace);262        last_pos = pos + search.length();263    }264    builder.append(s, last_pos, std::string::npos);265    s = std::move(builder);266}267 268static std::string gguf_kv_to_str(const struct gguf_context * ctx_gguf, int i) {269    const enum gguf_type type = gguf_get_kv_type(ctx_gguf, i);270 271    switch (type) {272        case GGUF_TYPE_STRING:273            return gguf_get_val_str(ctx_gguf, i);274        case GGUF_TYPE_ARRAY:275            {276                const enum gguf_type arr_type = gguf_get_arr_type(ctx_gguf, i);277                int arr_n = gguf_get_arr_n(ctx_gguf, i);278                const void * data = arr_type == GGUF_TYPE_STRING ? nullptr : gguf_get_arr_data(ctx_gguf, i);279                std::stringstream ss;280                ss << "[";281                for (int j = 0; j < arr_n; j++) {282                    if (arr_type == GGUF_TYPE_STRING) {283                        std::string val = gguf_get_arr_str(ctx_gguf, i, j);284                        // escape quotes285                        replace_all(val, "\\", "\\\\");286                        replace_all(val, "\"", "\\\"");287                        ss << '"' << val << '"';288                    } else if (arr_type == GGUF_TYPE_ARRAY) {289                        ss << "???";290                    } else {291                        ss << gguf_data_to_str(arr_type, data, j);292                    }293                    if (j < arr_n - 1) {294                        ss << ", ";295                    }296                }297                ss << "]";298                return ss.str();299            }300        default:301            return gguf_data_to_str(type, gguf_get_val_data(ctx_gguf, i), 0);302    }303}304 305static void print_tensor_info(const ggml_tensor * tensor, const char * prefix = "") {306    size_t tensor_size = ggml_nbytes(tensor);307    LOG_INF("%s: n_dims = %d, name = %s, tensor_size=%zu, shape:[%" PRId64 ", %" PRId64 ", %" PRId64 ", %" PRId64 "], type = %s\n",308            prefix, ggml_n_dims(tensor), tensor->name, tensor_size,309            tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3], ggml_type_name(tensor->type));310}311 312static projector_type clip_projector_type_from_string(const std::string & name) {313    for (const auto & kv : PROJECTOR_TYPE_NAMES) { // NOLINT314        if (kv.second == name) {315            return kv.first;316        }317    }318    return PROJECTOR_TYPE_UNKNOWN;319}320 321#ifdef CLIP_DEBUG_FUNCTIONS322static void clip_image_write_image_to_ppm(const clip_image_u8& img, const std::string& filename) {323    std::ofstream file(filename, std::ios::binary);324    if (!file.is_open()) {325        LOG_ERR("Failed to open file for writing: %s\n", filename.c_str());326        return;327    }328 329    // PPM header: P6 format, width, height, and max color value330    file << "P6\n" << img.nx << " " << img.ny << "\n255\n";331 332    // Write pixel data333    for (size_t i = 0; i < img.buf.size(); i += 3) {334        // PPM expects binary data in RGB format, which matches our image buffer335        file.write(reinterpret_cast<const char*>(&img.buf[i]), 3);336    }337 338    file.close();339}340 341static void clip_image_save_to_bmp(const clip_image_u8& img, const std::string& filename) {342    std::ofstream file(filename, std::ios::binary);343    if (!file.is_open()) {344        LOG_ERR("Failed to open file for writing: %s\n", filename.c_str());345        return;346    }347 348    int fileSize = 54 + 3 * img.nx * img.ny; // File header + info header + pixel data349    int bytesPerPixel = 3;350    int widthInBytes = img.nx * bytesPerPixel;351    int paddingAmount = (4 - (widthInBytes % 4)) % 4;352    int stride = widthInBytes + paddingAmount;353 354    // Bitmap file header355    unsigned char fileHeader[14] = {356        'B','M',     // Signature357        0,0,0,0,    // Image file size in bytes358        0,0,0,0,    // Reserved359        54,0,0,0    // Start of pixel array360    };361 362    // Total file size363    fileSize = 54 + (stride * img.ny);364    fileHeader[2] = (unsigned char)(fileSize);365    fileHeader[3] = (unsigned char)(fileSize >> 8);366    fileHeader[4] = (unsigned char)(fileSize >> 16);367    fileHeader[5] = (unsigned char)(fileSize >> 24);368 369    // Bitmap information header (BITMAPINFOHEADER)370    unsigned char infoHeader[40] = {371        40,0,0,0,   // Size of this header (40 bytes)372        0,0,0,0,    // Image width373        0,0,0,0,    // Image height374        1,0,        // Number of color planes375        24,0,       // Bits per pixel376        0,0,0,0,    // No compression377        0,0,0,0,    // Image size (can be 0 for no compression)378        0,0,0,0,    // X pixels per meter (not specified)379        0,0,0,0,    // Y pixels per meter (not specified)380        0,0,0,0,    // Total colors (color table not used)381        0,0,0,0     // Important colors (all are important)382    };383 384    // Width and height in the information header385    infoHeader[4] = (unsigned char)(img.nx);386    infoHeader[5] = (unsigned char)(img.nx >> 8);387    infoHeader[6] = (unsigned char)(img.nx >> 16);388    infoHeader[7] = (unsigned char)(img.nx >> 24);389    infoHeader[8] = (unsigned char)(img.ny);390    infoHeader[9] = (unsigned char)(img.ny >> 8);391    infoHeader[10] = (unsigned char)(img.ny >> 16);392    infoHeader[11] = (unsigned char)(img.ny >> 24);393 394    // Write file headers395    file.write(reinterpret_cast<char*>(fileHeader), sizeof(fileHeader));396    file.write(reinterpret_cast<char*>(infoHeader), sizeof(infoHeader));397 398    // Pixel data399    std::vector<unsigned char> padding(3, 0); // Max padding size to be added to each row400    for (int y = img.ny - 1; y >= 0; --y) { // BMP files are stored bottom-to-top401        for (int x = 0; x < img.nx; ++x) {402            // Each pixel403            size_t pixelIndex = (y * img.nx + x) * 3;404            unsigned char pixel[3] = {405                img.buf[pixelIndex + 2], // BMP stores pixels in BGR format406                img.buf[pixelIndex + 1],407                img.buf[pixelIndex]408            };409            file.write(reinterpret_cast<char*>(pixel), 3);410        }411        // Write padding for the row412        file.write(reinterpret_cast<char*>(padding.data()), paddingAmount);413    }414 415    file.close();416}417 418// debug function to convert f32 to u8419static void clip_image_convert_f32_to_u8(const clip_image_f32& src, clip_image_u8& dst) {420    dst.nx = src.nx;421    dst.ny = src.ny;422    dst.buf.resize(3 * src.nx * src.ny);423    for (size_t i = 0; i < src.buf.size(); ++i) {424        dst.buf[i] = static_cast<uint8_t>(std::min(std::max(int(src.buf[i] * 255.0f), 0), 255));425    }426}427#endif428 429 430//431// clip layers432//433 434struct clip_hparams {435    int32_t image_size;436    int32_t patch_size;437    int32_t hidden_size;438    int32_t n_intermediate;439    int32_t projection_dim;440    int32_t n_head;441    int32_t n_layer;442 443    float eps;444 445    char mm_patch_merge_type[32] = "flat"; // spatial_unpad or flat (default)446 447    int32_t image_grid_pinpoints[32];448    int32_t image_crop_resolution;449};450 451struct clip_layer {452    // attention453    struct ggml_tensor * k_w;454    struct ggml_tensor * k_b;455    struct ggml_tensor * q_w;456    struct ggml_tensor * q_b;457    struct ggml_tensor * v_w;458    struct ggml_tensor * v_b;459 460    struct ggml_tensor * o_w;461    struct ggml_tensor * o_b;462 463    // layernorm 1464    struct ggml_tensor * ln_1_w;465    struct ggml_tensor * ln_1_b;466 467    // ff468    struct ggml_tensor * ff_i_w;469    struct ggml_tensor * ff_i_b;470 471    struct ggml_tensor * ff_o_w;472    struct ggml_tensor * ff_o_b;473 474    // layernorm 2475    struct ggml_tensor * ln_2_w;476    struct ggml_tensor * ln_2_b;477};478 479struct clip_vision_model {480    struct clip_hparams hparams;481 482    // embeddings483    struct ggml_tensor * class_embedding;484    struct ggml_tensor * patch_embeddings_0;485    struct ggml_tensor * patch_embeddings_1;  // second Conv2D kernel when we decouple Conv3D along temproal dimension (Qwen2VL)486    struct ggml_tensor * patch_bias;487    struct ggml_tensor * position_embeddings;488 489    struct ggml_tensor * pre_ln_w;490    struct ggml_tensor * pre_ln_b;491 492    std::vector<clip_layer> layers;493 494    struct ggml_tensor * post_ln_w;495    struct ggml_tensor * post_ln_b;496 497    struct ggml_tensor * projection;498 499    // LLaVA projection500    struct ggml_tensor * mm_0_w = NULL;501    struct ggml_tensor * mm_0_b = NULL;502    struct ggml_tensor * mm_2_w = NULL;503    struct ggml_tensor * mm_2_b = NULL;504 505    struct ggml_tensor * image_newline = NULL;506 507    // Yi type models with mlp+normalization projection508    struct ggml_tensor * mm_1_w = NULL; // Yi type models have 0, 1, 3, 4509    struct ggml_tensor * mm_1_b = NULL;510    struct ggml_tensor * mm_3_w = NULL;511    struct ggml_tensor * mm_3_b = NULL;512    struct ggml_tensor * mm_4_w = NULL;513    struct ggml_tensor * mm_4_b = NULL;514 515    //GLMV-Edge projection516    struct ggml_tensor * mm_model_adapter_conv_w;517    struct ggml_tensor * mm_model_adapter_conv_b;518    struct ggml_tensor * boi_w;519    struct ggml_tensor * eoi_w;520 521    // MobileVLM projection522    struct ggml_tensor * mm_model_mlp_1_w;523    struct ggml_tensor * mm_model_mlp_1_b;524    struct ggml_tensor * mm_model_mlp_3_w;525    struct ggml_tensor * mm_model_mlp_3_b;526    struct ggml_tensor * mm_model_block_1_block_0_0_w;527    struct ggml_tensor * mm_model_block_1_block_0_1_w;528    struct ggml_tensor * mm_model_block_1_block_0_1_b;529    struct ggml_tensor * mm_model_block_1_block_1_fc1_w;530    struct ggml_tensor * mm_model_block_1_block_1_fc1_b;531    struct ggml_tensor * mm_model_block_1_block_1_fc2_w;532    struct ggml_tensor * mm_model_block_1_block_1_fc2_b;533    struct ggml_tensor * mm_model_block_1_block_2_0_w;534    struct ggml_tensor * mm_model_block_1_block_2_1_w;535    struct ggml_tensor * mm_model_block_1_block_2_1_b;536    struct ggml_tensor * mm_model_block_2_block_0_0_w;537    struct ggml_tensor * mm_model_block_2_block_0_1_w;538    struct ggml_tensor * mm_model_block_2_block_0_1_b;539    struct ggml_tensor * mm_model_block_2_block_1_fc1_w;540    struct ggml_tensor * mm_model_block_2_block_1_fc1_b;541    struct ggml_tensor * mm_model_block_2_block_1_fc2_w;542    struct ggml_tensor * mm_model_block_2_block_1_fc2_b;543    struct ggml_tensor * mm_model_block_2_block_2_0_w;544    struct ggml_tensor * mm_model_block_2_block_2_1_w;545    struct ggml_tensor * mm_model_block_2_block_2_1_b;546 547    // MobileVLM_V2 projection548    struct ggml_tensor * mm_model_mlp_0_w;549    struct ggml_tensor * mm_model_mlp_0_b;550    struct ggml_tensor * mm_model_mlp_2_w;551    struct ggml_tensor * mm_model_mlp_2_b;552    struct ggml_tensor * mm_model_peg_0_w;553    struct ggml_tensor * mm_model_peg_0_b;554 555    // MINICPMV projection556    struct ggml_tensor * mm_model_pos_embed_k;557    struct ggml_tensor * mm_model_query;558    struct ggml_tensor * mm_model_proj;559    struct ggml_tensor * mm_model_kv_proj;560    struct ggml_tensor * mm_model_attn_q_w;561    struct ggml_tensor * mm_model_attn_q_b;562    struct ggml_tensor * mm_model_attn_k_w;563    struct ggml_tensor * mm_model_attn_k_b;564    struct ggml_tensor * mm_model_attn_v_w;565    struct ggml_tensor * mm_model_attn_v_b;566    struct ggml_tensor * mm_model_attn_o_w;567    struct ggml_tensor * mm_model_attn_o_b;568    struct ggml_tensor * mm_model_ln_q_w;569    struct ggml_tensor * mm_model_ln_q_b;570    struct ggml_tensor * mm_model_ln_kv_w;571    struct ggml_tensor * mm_model_ln_kv_b;572    struct ggml_tensor * mm_model_ln_post_w;573    struct ggml_tensor * mm_model_ln_post_b;574};575 576struct clip_ctx {577    bool has_text_encoder    = false;578    bool has_vision_encoder  = false;579    bool has_llava_projector = false;580    bool has_minicpmv_projector = false;581    bool has_glm_projector = false;582    bool has_qwen2vl_merger = false;583    int minicpmv_version = 2;584 585    struct clip_vision_model vision_model;586    projector_type proj_type = PROJECTOR_TYPE_MLP;587 588    float image_mean[3];589    float image_std[3];590    bool use_gelu = false;591    bool use_silu = false;592    int32_t ftype = 1;593 594    bool has_class_embedding = true;595    bool has_pre_norm = true;596    bool has_post_norm = false;597    bool has_patch_bias = false;598 599    struct gguf_context * ctx_gguf;600    struct ggml_context * ctx_data;601 602    std::vector<uint8_t> buf_compute_meta;603 604    // memory buffers to evaluate the model605    ggml_backend_buffer_t params_buffer  = NULL;606 607    ggml_backend_t backend       = NULL;608    ggml_gallocr_t compute_alloc = NULL;609 610    struct clip_image_size * load_image_size;611};612 613static ggml_cgraph * clip_image_build_graph(clip_ctx * ctx, const clip_image_f32_batch * imgs, struct clip_image_size * load_image_size, bool is_inf = false) {614    if (!ctx->has_vision_encoder) {615        LOG_ERR("This gguf file seems to have no vision encoder\n");616        return nullptr;617    }618 619    const auto & model = ctx->vision_model;620    const auto & hparams = model.hparams;621 622    const int image_size = hparams.image_size;623    int image_size_width  = image_size;624    int image_size_height = image_size;625    if (ctx->has_minicpmv_projector) {626        if (load_image_size == nullptr) {627            load_image_size = clip_image_size_init();628        }629        LOG_DBG("%s: %d %d\n", __func__, load_image_size->width, load_image_size->height);630        image_size_width  = load_image_size->width;631        image_size_height = load_image_size->height;632        if (is_inf) {633            image_size_width  = imgs->data->nx;634            image_size_height = imgs->data->ny;635        }636    }637    else if (ctx->has_qwen2vl_merger) {638        // use the image's native resolution when image is avaible639        if (is_inf) {640        // if (imgs->data->nx && imgs->data->ny) {641            image_size_width  = imgs->data->nx;642            image_size_height = imgs->data->ny;643        }644    }645    const int patch_size           = hparams.patch_size;646    const int num_patches          = ((image_size_width / patch_size) * (image_size_height / patch_size));647    const int patches_w            = image_size_width / patch_size;648    const int patches_h            = image_size_height / patch_size;649    const int num_positions        = num_patches + (ctx->has_class_embedding ? 1 : 0);650    const int num_position_ids     = ctx->has_qwen2vl_merger ? num_positions * 4 : num_positions;651    const int hidden_size          = hparams.hidden_size;652    const int n_head               = hparams.n_head;653    const int d_head               = hidden_size / n_head;654    int n_layer                    = hparams.n_layer;655    const float eps                = hparams.eps;656    int mrope_sections[4] = {d_head/4, d_head/4, d_head/4, d_head/4};657 658    const int batch_size = imgs->size;659 660    if (ctx->has_llava_projector || ctx->has_minicpmv_projector || ctx->has_glm_projector) {661        GGML_ASSERT(batch_size == 1);662    }663 664    struct ggml_init_params params = {665        /*.mem_size   =*/ ctx->buf_compute_meta.size(),666        /*.mem_buffer =*/ ctx->buf_compute_meta.data(),667        /*.no_alloc   =*/ true,668    };669 670    struct ggml_context * ctx0 = ggml_init(params);671    struct ggml_cgraph * gf = ggml_new_graph(ctx0);672 673    struct ggml_tensor * inp_raw = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, image_size_width, image_size_height, 3, batch_size);674    ggml_set_name(inp_raw, "inp_raw");675    ggml_set_input(inp_raw);676 677    struct ggml_tensor * inp = ggml_conv_2d(ctx0, model.patch_embeddings_0, inp_raw, patch_size, patch_size, 0, 0, 1, 1);678 679    if (ctx->has_qwen2vl_merger) {680        GGML_ASSERT(image_size_width % (patch_size * 2) == 0);681        GGML_ASSERT(image_size_height % (patch_size * 2) == 0);682 683        auto inp_1 = ggml_conv_2d(ctx0, model.patch_embeddings_1, inp_raw, patch_size, patch_size, 0, 0, 1, 1);684        inp = ggml_add(ctx0, inp, inp_1);685        inp = ggml_cont(ctx0, ggml_permute(ctx0, inp, 1, 2, 0, 3));  // [w, h, c, b] -> [c, w, h, b]686        inp = ggml_reshape_4d(687            ctx0, inp,688            hidden_size * 2, patches_w / 2, patches_h, batch_size);689        inp = ggml_reshape_4d(690            ctx0, inp,691            hidden_size * 2, patches_w / 2, 2, batch_size * (patches_h / 2));692        inp = ggml_cont(ctx0, ggml_permute(ctx0, inp, 0, 2, 1, 3));693        inp = ggml_reshape_3d(694            ctx0, inp,695            hidden_size, patches_w * patches_h, batch_size);696    }697    else {698        inp = ggml_reshape_3d(ctx0, inp, num_patches, hidden_size, batch_size);699        inp = ggml_cont(ctx0, ggml_permute(ctx0, inp, 1, 0, 2, 3));700    }701 702    if (ctx->has_patch_bias) {703        // inp = ggml_add(ctx0, inp, ggml_repeat(ctx0, model.patch_bias, inp));704        inp = ggml_add(ctx0, inp, model.patch_bias);705    }706    struct ggml_tensor * embeddings = inp;707    struct ggml_tensor * pos_embed = nullptr;708 709    if (ctx->has_llava_projector) {710        // concat class_embeddings and patch_embeddings711        if (ctx->has_class_embedding) {712            embeddings = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, hidden_size, num_positions, batch_size);713            ggml_set_name(embeddings, "embeddings");714            ggml_set_input(embeddings);715            embeddings = ggml_acc(ctx0, embeddings, model.class_embedding,716                    embeddings->nb[1], embeddings->nb[2], embeddings->nb[3], 0);717            embeddings = ggml_acc(ctx0, embeddings, inp,718                    embeddings->nb[1], embeddings->nb[2], embeddings->nb[3], model.class_embedding->nb[1]);719        }720    }721 722    struct ggml_tensor * positions = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, num_position_ids);723    ggml_set_name(positions, "positions");724    ggml_set_input(positions);725 726    if (!ctx->has_qwen2vl_merger) { // qwen2vl use rope position embedding727        embeddings =728            ggml_add(ctx0, embeddings, ggml_get_rows(ctx0, model.position_embeddings, positions));729    }730 731    if (ctx->has_minicpmv_projector) {732        int pos_w = image_size_width/patch_size;733        int pos_h = image_size_height/patch_size;734        if (ctx->minicpmv_version == 2) {735            pos_embed = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 4096, pos_w * pos_h, 1);736        }737        else if (ctx->minicpmv_version == 3) {738            pos_embed = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 3584, pos_w * pos_h, 1);739        }740        else if (ctx->minicpmv_version == 4) {741            pos_embed = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 3584, pos_w * pos_h, 1);742        }743        ggml_set_name(pos_embed, "pos_embed");744        ggml_set_input(pos_embed);745    }746 747    // pre-layernorm748    if (ctx->has_pre_norm) {749        embeddings = ggml_norm(ctx0, embeddings, eps);750        ggml_set_name(embeddings, "pre_ln");751 752        embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.pre_ln_w), model.pre_ln_b);753    }754 755    // loop over layers756    if (ctx->has_minicpmv_projector || ctx->has_glm_projector || ctx->has_qwen2vl_merger) {757        n_layer += 1;758    }759    for (int il = 0; il < n_layer - 1; il++) {760        struct ggml_tensor * cur = embeddings; // embeddings = residual, cur = hidden_states761 762        //const size_t nb_q_w = model.layers[il].q_w->nb[0];763 764        // layernorm1765        {766            cur = ggml_norm(ctx0, cur, eps);767 768            cur = ggml_add(ctx0, ggml_mul(ctx0, cur, model.layers[il].ln_1_w),769                           model.layers[il].ln_1_b);770        }771 772        // self-attention773        {774 775            struct ggml_tensor * Q =776                ggml_add(ctx0, ggml_mul_mat(ctx0, model.layers[il].q_w, cur), model.layers[il].q_b);777 778            Q = ggml_reshape_4d(ctx0, Q, d_head, n_head, num_positions, batch_size);779            if (ctx->has_qwen2vl_merger) {780                Q = ggml_rope_multi(781                    ctx0, Q, positions, nullptr,782                    d_head/2, mrope_sections, GGML_ROPE_TYPE_VISION, 32768, 10000, 1, 0, 1, 32, 1);783            }784            Q = ggml_scale_inplace(ctx0, Q, 1.0f / sqrt((float)d_head));785            Q = ggml_cont(ctx0, ggml_permute(ctx0, Q, 0, 2, 1, 3));786            Q = ggml_reshape_3d(ctx0, Q, d_head, num_positions, n_head * batch_size);787 788            struct ggml_tensor * K =789                ggml_add(ctx0, ggml_mul_mat(ctx0, model.layers[il].k_w, cur), model.layers[il].k_b);790 791            K = ggml_reshape_4d(ctx0, K, d_head, n_head, num_positions, batch_size);792            if (ctx->has_qwen2vl_merger) {793                K = ggml_rope_multi(794                    ctx0, K, positions, nullptr,795                    d_head/2, mrope_sections, GGML_ROPE_TYPE_VISION, 32768, 10000, 1, 0, 1, 32, 1);796            }797            K = ggml_cont(ctx0, ggml_permute(ctx0, K, 0, 2, 1, 3));798            K = ggml_reshape_3d(ctx0, K, d_head, num_positions, n_head * batch_size);799 800            struct ggml_tensor * V =801                ggml_add(ctx0, ggml_mul_mat(ctx0, model.layers[il].v_w, cur), model.layers[il].v_b);802 803            V = ggml_reshape_4d(ctx0, V, d_head, n_head, num_positions, batch_size);804            V = ggml_cont(ctx0, ggml_permute(ctx0, V, 1, 2, 0, 3));805            V = ggml_reshape_3d(ctx0, V, num_positions, d_head, n_head * batch_size);806 807            struct ggml_tensor * KQ = ggml_mul_mat(ctx0, K, Q);808            KQ = ggml_soft_max_inplace(ctx0, KQ);809            struct ggml_tensor * KQV = ggml_mul_mat(ctx0, V, KQ);810            KQV = ggml_reshape_4d(ctx0, KQV, d_head, num_positions, n_head, batch_size);811            KQV = ggml_permute(ctx0, KQV, 0, 2, 1, 3);812 813            cur = ggml_cont_3d(ctx0, KQV, hidden_size, num_positions, batch_size);814        }815 816        // attention output817        cur = ggml_add(ctx0, ggml_mul_mat(ctx0, model.layers[il].o_w, cur), model.layers[il].o_b);818 819        // re-add the layer input, e.g., residual820        cur = ggml_add(ctx0, cur, embeddings);821 822        embeddings = cur; // embeddings = residual, cur = hidden_states823 824        // layernorm2825        {826            cur = ggml_norm(ctx0, cur, eps);827 828            cur = ggml_add(ctx0, ggml_mul(ctx0, cur, model.layers[il].ln_2_w), model.layers[il].ln_2_b);829        }830 831        cur = ggml_mul_mat(ctx0, model.layers[il].ff_i_w, cur);832        cur = ggml_add(ctx0, cur, model.layers[il].ff_i_b);833 834        if (ctx->use_gelu) {835            cur = ggml_gelu_inplace(ctx0, cur);836        } else if (ctx->use_silu) {837            cur = ggml_silu_inplace(ctx0, cur);838        } else {839            cur = ggml_gelu_quick_inplace(ctx0, cur);840        }841 842        cur = ggml_mul_mat(ctx0, model.layers[il].ff_o_w, cur);843        cur = ggml_add(ctx0, cur, model.layers[il].ff_o_b);844 845        // residual 2846        cur = ggml_add(ctx0, embeddings, cur);847 848        embeddings = cur;849 850    }851 852    // post-layernorm853    if (ctx->has_post_norm) {854        embeddings = ggml_norm(ctx0, embeddings, eps);855        ggml_set_name(embeddings, "post_ln");856 857        embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.post_ln_w), model.post_ln_b);858    }859 860    // llava projector861    if (ctx->has_llava_projector) {862        embeddings = ggml_reshape_2d(ctx0, embeddings, embeddings->ne[0], embeddings->ne[1]);863 864        struct ggml_tensor * patches = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, num_patches);865        ggml_set_name(patches, "patches");866        ggml_set_input(patches);867 868        // shape [1, 576, 1024]869        // ne is whcn, ne = [1024, 576, 1, 1]870        embeddings = ggml_get_rows(ctx0, embeddings, patches);871 872        // print_tensor_info(embeddings, "embeddings");873 874        // llava projector875        if (ctx->proj_type == PROJECTOR_TYPE_MLP) {876            embeddings = ggml_mul_mat(ctx0, model.mm_0_w, embeddings);877            embeddings = ggml_add(ctx0, embeddings, model.mm_0_b);878 879            embeddings = ggml_gelu(ctx0, embeddings);880            embeddings = ggml_mul_mat(ctx0, model.mm_2_w, embeddings);881            embeddings = ggml_add(ctx0, embeddings, model.mm_2_b);882        }883        else if (ctx->proj_type == PROJECTOR_TYPE_MLP_NORM) {884            embeddings = ggml_mul_mat(ctx0, model.mm_0_w, embeddings);885            embeddings = ggml_add(ctx0, embeddings, model.mm_0_b);886            // ggml_tensor_printf(embeddings, "mm_0_w",0,true,false);887            // First LayerNorm888            embeddings = ggml_norm(ctx0, embeddings, eps);889            embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_1_w),890                                model.mm_1_b);891 892            // GELU activation893            embeddings = ggml_gelu(ctx0, embeddings);894 895            // Second linear layer896            embeddings = ggml_mul_mat(ctx0, model.mm_3_w, embeddings);897            embeddings = ggml_add(ctx0, embeddings, model.mm_3_b);898 899            // Second LayerNorm900            embeddings = ggml_norm(ctx0, embeddings, eps);901            embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_4_w),902                                model.mm_4_b);903        }904        else if (ctx->proj_type == PROJECTOR_TYPE_LDP) {905            // MobileVLM projector906            int n_patch = 24;907            struct ggml_tensor * mlp_1 = ggml_mul_mat(ctx0, model.mm_model_mlp_1_w, embeddings);908            mlp_1 = ggml_add(ctx0, mlp_1, model.mm_model_mlp_1_b);909            mlp_1 = ggml_gelu(ctx0, mlp_1);910            struct ggml_tensor * mlp_3 = ggml_mul_mat(ctx0, model.mm_model_mlp_3_w, mlp_1);911            mlp_3 = ggml_add(ctx0, mlp_3, model.mm_model_mlp_3_b);912            // mlp_3 shape = [1, 576, 2048], ne = [2048, 576, 1, 1]913 914            // block 1915            struct ggml_tensor * block_1 = nullptr;916            {917                // transpose from [1, 576, 2048] --> [1, 2048, 576] --> [1, 2048, 24, 24]918                mlp_3 = ggml_cont(ctx0, ggml_permute(ctx0, mlp_3, 1, 0, 2, 3));919                mlp_3 = ggml_reshape_4d(ctx0, mlp_3, n_patch, n_patch, mlp_3->ne[1], mlp_3->ne[2]);920                // stride = 1, padding = 1, bias is nullptr921                block_1 = ggml_conv_2d_dw(ctx0, model.mm_model_block_1_block_0_0_w, mlp_3, 1, 1, 1, 1, 1, 1);922 923                // layer norm924                // // block_1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]925                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 2, 0, 3));926                // block_1 shape = [1, 24, 24, 2048], ne = [2048, 24, 24, 1]927                block_1 = ggml_norm(ctx0, block_1, eps);928                block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_1_block_0_1_w), model.mm_model_block_1_block_0_1_b);929                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));930 931                // block_1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]932                // hardswish933                struct ggml_tensor * block_1_hw = ggml_hardswish(ctx0, block_1);934 935                block_1 = ggml_pool_2d(ctx0, block_1_hw, GGML_OP_POOL_AVG, block_1_hw->ne[0], block_1_hw->ne[1], block_1_hw->ne[0], block_1_hw->ne[1], 0, 0);936                // block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]937                // pointwise conv938                block_1 = ggml_reshape_2d(ctx0, block_1, block_1->ne[0]*block_1->ne[1]*block_1->ne[2], block_1->ne[3]);939                block_1 = ggml_mul_mat(ctx0, model.mm_model_block_1_block_1_fc1_w, block_1);940                block_1 = ggml_add(ctx0, block_1, model.mm_model_block_1_block_1_fc1_b);941                block_1 = ggml_relu(ctx0, block_1);942                block_1 = ggml_mul_mat(ctx0, model.mm_model_block_1_block_1_fc2_w, block_1);943                block_1 = ggml_add(ctx0, block_1, model.mm_model_block_1_block_1_fc2_b);944                block_1 = ggml_hardsigmoid(ctx0, block_1);945                // block_1_hw shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1], block_1 shape = [1, 2048], ne = [2048, 1, 1, 1]946                block_1 = ggml_reshape_4d(ctx0, block_1, 1, 1, block_1->ne[0], block_1->ne[1]);947                block_1 = ggml_mul(ctx0, block_1_hw, block_1);948 949                int w = block_1->ne[0], h = block_1->ne[1];950                block_1 = ggml_reshape_3d(ctx0, block_1, w*h, block_1->ne[2], block_1->ne[3]);951                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 0, 2, 3));952 953                // block_1 shape = [1, 24*24, 2048], ne = [24*24, 2048, 1]954                block_1 = ggml_mul_mat(ctx0, model.mm_model_block_1_block_2_0_w, block_1);955                block_1 = ggml_reshape_4d(ctx0, block_1, block_1->ne[0], w, h, block_1->ne[3]);956 957                // block_1 shape = [1, 24, 24, 2048], ne = [2048, 24, 24, 1]958                block_1 = ggml_norm(ctx0, block_1, eps);959                block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_1_block_2_1_w), model.mm_model_block_1_block_2_1_b);960                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));961                // block1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]962                // residual963                block_1 = ggml_add(ctx0, mlp_3, block_1);964            }965 966            // block_2967            {968                // stride = 2969                block_1 = ggml_conv_2d_dw(ctx0, model.mm_model_block_2_block_0_0_w, block_1, 2, 2, 1, 1, 1, 1);970 971                // block_1 shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1]972                // layer norm973                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 2, 0, 3));974                // block_1 shape = [1, 12, 12, 2048], ne = [2048, 12, 12, 1]975                block_1 = ggml_norm(ctx0, block_1, eps);976                block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_2_block_0_1_w), model.mm_model_block_2_block_0_1_b);977                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));978                // block_1 shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1]979                // hardswish980                struct ggml_tensor * block_1_hw = ggml_hardswish(ctx0, block_1);981 982                // not sure the parameters is right for globalAvgPooling983                block_1 = ggml_pool_2d(ctx0, block_1_hw, GGML_OP_POOL_AVG, block_1_hw->ne[0], block_1_hw->ne[1], block_1_hw->ne[0], block_1_hw->ne[1], 0, 0);984                // block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]985                // pointwise conv986                block_1 = ggml_reshape_2d(ctx0, block_1, block_1->ne[0]*block_1->ne[1]*block_1->ne[2], block_1->ne[3]);987                block_1 = ggml_mul_mat(ctx0, model.mm_model_block_2_block_1_fc1_w, block_1);988                block_1 = ggml_add(ctx0, block_1, model.mm_model_block_2_block_1_fc1_b);989                block_1 = ggml_relu(ctx0, block_1);990                block_1 = ggml_mul_mat(ctx0, model.mm_model_block_2_block_1_fc2_w, block_1);991                block_1 = ggml_add(ctx0, block_1, model.mm_model_block_2_block_1_fc2_b);992                block_1 = ggml_hardsigmoid(ctx0, block_1);993 994                // block_1_hw shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1], block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]995                block_1 = ggml_reshape_4d(ctx0, block_1, 1, 1, block_1->ne[0], block_1->ne[1]);996                block_1 = ggml_mul(ctx0, block_1_hw, block_1);997 998                int w = block_1->ne[0], h = block_1->ne[1];999                block_1 = ggml_reshape_3d(ctx0, block_1, w*h, block_1->ne[2], block_1->ne[3]);1000                block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 0, 2, 3));1001                // block_1 shape = [1, 24*24, 2048], ne = [24*24, 2048, 1]1002                block_1 = ggml_mul_mat(ctx0, model.mm_model_block_2_block_2_0_w, block_1);1003                block_1 = ggml_reshape_4d(ctx0, block_1, block_1->ne[0], w, h, block_1->ne[3]);1004 1005 1006                // block_1 shape = [1, 12, 12, 2048], ne = [2048, 12, 12, 1]1007                block_1 = ggml_norm(ctx0, block_1, eps);1008                block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_2_block_2_1_w), model.mm_model_block_2_block_2_1_b);1009                block_1 = ggml_reshape_3d(ctx0, block_1, block_1->ne[0], block_1->ne[1] * block_1->ne[2], block_1->ne[3]);1010                // block_1 shape = [1, 144, 2048], ne = [2048, 144, 1]1011            }1012            embeddings = block_1;1013        }1014        else if (ctx->proj_type == PROJECTOR_TYPE_LDPV2)1015        {1016            int n_patch = 24;1017            struct ggml_tensor * mlp_0 = ggml_mul_mat(ctx0, model.mm_model_mlp_0_w, embeddings);1018            mlp_0 = ggml_add(ctx0, mlp_0, model.mm_model_mlp_0_b);1019            mlp_0 = ggml_gelu(ctx0, mlp_0);1020            struct ggml_tensor * mlp_2 = ggml_mul_mat(ctx0, model.mm_model_mlp_2_w, mlp_0);1021            mlp_2 = ggml_add(ctx0, mlp_2, model.mm_model_mlp_2_b);1022            // mlp_2 ne = [2048, 576, 1, 1]1023            // // AVG Pool Layer 2*2, strides = 21024            mlp_2 = ggml_cont(ctx0, ggml_permute(ctx0, mlp_2, 1, 0, 2, 3));1025            // mlp_2 ne = [576, 2048, 1, 1]1026            mlp_2 = ggml_reshape_4d(ctx0, mlp_2, n_patch, n_patch, mlp_2->ne[1], mlp_2->ne[2]);1027            // mlp_2 ne [24, 24, 2048, 1]1028            mlp_2 = ggml_pool_2d(ctx0, mlp_2, GGML_OP_POOL_AVG, 2, 2, 2, 2, 0, 0);1029            // weight ne = [3, 3, 2048, 1]1030            struct ggml_tensor * peg_0 = ggml_conv_2d_dw(ctx0, model.mm_model_peg_0_w, mlp_2, 1, 1, 1, 1, 1, 1);1031            peg_0 = ggml_cont(ctx0, ggml_permute(ctx0, peg_0, 1, 2, 0, 3));1032            peg_0 = ggml_add(ctx0, peg_0, model.mm_model_peg_0_b);1033            mlp_2 = ggml_cont(ctx0, ggml_permute(ctx0, mlp_2, 1, 2, 0, 3));1034            peg_0 = ggml_add(ctx0, peg_0, mlp_2);1035            peg_0 = ggml_reshape_3d(ctx0, peg_0, peg_0->ne[0], peg_0->ne[1] * peg_0->ne[2], peg_0->ne[3]);1036            embeddings = peg_0;1037        }1038        else {1039            GGML_ABORT("fatal error");1040        }1041    }1042    // minicpmv projector1043    else if (ctx->has_minicpmv_projector)1044    {1045        if (ctx->proj_type == PROJECTOR_TYPE_RESAMPLER) {1046            struct ggml_tensor * q = model.mm_model_query;1047            { // layernorm1048                q = ggml_norm(ctx0, q, eps);1049                q = ggml_add(ctx0, ggml_mul(ctx0, q, model.mm_model_ln_q_w), model.mm_model_ln_q_b);1050            }1051            struct ggml_tensor * v = ggml_mul_mat(ctx0, model.mm_model_kv_proj, embeddings);1052            { // layernorm1053                v = ggml_norm(ctx0, v, eps);1054                v = ggml_add(ctx0, ggml_mul(ctx0, v, model.mm_model_ln_kv_w), model.mm_model_ln_kv_b);1055            }1056            struct ggml_tensor * k;1057            { // position1058                // q = ggml_add(ctx0, q, model.mm_model_pos_embed);1059                k = ggml_add(ctx0, v, pos_embed);1060            }1061 1062            { // attention1063                int hidden_size = 4096;1064                const int d_head = 128;1065                int n_head = hidden_size/d_head;1066                int num_query = 96;1067                if (ctx->minicpmv_version == 2) {1068                    hidden_size = 4096;1069                    n_head = hidden_size/d_head;1070                    num_query = 96;1071                }1072                else if (ctx->minicpmv_version == 3) {1073                    hidden_size = 3584;1074                    n_head = hidden_size/d_head;1075                    num_query = 64;1076                }1077                else if (ctx->minicpmv_version == 4) {1078                    hidden_size = 3584;1079                    n_head = hidden_size/d_head;1080                    num_query = 64;1081                }1082 1083                struct ggml_tensor * Q = ggml_add(ctx0, ggml_mul_mat(ctx0, model.mm_model_attn_q_w, q), model.mm_model_attn_q_b);1084                Q = ggml_scale_inplace(ctx0, Q, 1.0f / sqrt((float)d_head));1085                struct ggml_tensor * K = ggml_add(ctx0, ggml_mul_mat(ctx0, model.mm_model_attn_k_w, k), model.mm_model_attn_k_b);1086                struct ggml_tensor * V = ggml_add(ctx0, ggml_mul_mat(ctx0, model.mm_model_attn_v_w, v), model.mm_model_attn_v_b);1087                // permute1088                Q = ggml_reshape_4d(ctx0, Q, d_head, n_head, num_query, batch_size);1089                Q = ggml_cont(ctx0, ggml_permute(ctx0, Q, 0, 2, 1, 3));1090                Q = ggml_reshape_3d(ctx0, Q, d_head, num_query, n_head * batch_size);1091                K = ggml_reshape_4d(ctx0, K, d_head, n_head, num_positions, batch_size);1092                K = ggml_cont(ctx0, ggml_permute(ctx0, K, 0, 2, 1, 3));1093                K = ggml_reshape_3d(ctx0, K, d_head, num_positions, n_head * batch_size);1094                V = ggml_reshape_4d(ctx0, V, d_head, n_head, num_positions, batch_size);1095                V = ggml_cont(ctx0, ggml_permute(ctx0, V, 1, 2, 0, 3));1096                V = ggml_reshape_3d(ctx0, V, num_positions, d_head, n_head * batch_size);1097                struct ggml_tensor * KQ = ggml_mul_mat(ctx0, K, Q);1098                KQ = ggml_soft_max_inplace(ctx0, KQ);1099                struct ggml_tensor * KQV = ggml_mul_mat(ctx0, V, KQ);1100                KQV = ggml_reshape_4d(ctx0, KQV, d_head, num_query, n_head, batch_size);1101                KQV = ggml_permute(ctx0, KQV, 0, 2, 1, 3);1102                KQV = ggml_cont_3d(ctx0, KQV, hidden_size, num_query, batch_size);1103 1104                embeddings = ggml_add(ctx0, ggml_mul_mat(ctx0, model.mm_model_attn_o_w, KQV), model.mm_model_attn_o_b);1105            }1106            { // layernorm1107                embeddings = ggml_norm(ctx0, embeddings, eps);1108                embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_model_ln_post_w), model.mm_model_ln_post_b);1109            }1110            embeddings = ggml_mul_mat(ctx0, model.mm_model_proj, embeddings);1111        }1112        else {1113            GGML_ASSERT(false);1114        }1115    }1116    // glm projector1117    else if (ctx->has_glm_projector) {1118        if (ctx->proj_type == PROJECTOR_TYPE_GLM_EDGE) {1119            size_t gridsz = (size_t)sqrt(embeddings->ne[1]);1120            embeddings = ggml_cont(ctx0, ggml_permute(ctx0,embeddings,1,0,2,3));1121            embeddings = ggml_reshape_3d(ctx0, embeddings, gridsz, gridsz, embeddings->ne[1]);1122            embeddings = ggml_conv_2d(ctx0, model.mm_model_adapter_conv_w, embeddings, 2, 2, 0, 0, 1, 1);1123            embeddings = ggml_reshape_3d(ctx0, embeddings,embeddings->ne[0]*embeddings->ne[1] , embeddings->ne[2], batch_size);1124            embeddings = ggml_cont(ctx0, ggml_permute(ctx0,embeddings, 1, 0, 2, 3));1125            embeddings = ggml_add(ctx0, embeddings, model.mm_model_adapter_conv_b);1126            //GLU1127            {1128                embeddings = ggml_mul_mat(ctx0, model.mm_model_mlp_0_w, embeddings);1129                embeddings = ggml_norm(ctx0, embeddings, eps);1130                embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_model_ln_q_w), model.mm_model_ln_q_b);1131                embeddings = ggml_gelu_inplace(ctx0, embeddings);1132                struct ggml_tensor * x = embeddings;1133                embeddings = ggml_mul_mat(ctx0, model.mm_model_mlp_2_w, embeddings);1134                x = ggml_mul_mat(ctx0, model.mm_model_mlp_1_w,x);1135                embeddings = ggml_silu_inplace(ctx0, embeddings);1136                embeddings = ggml_mul(ctx0, embeddings,x);1137                embeddings = ggml_mul_mat(ctx0, model.mm_model_mlp_3_w, embeddings);1138            }1139        } else {1140            GGML_ABORT("fatel error");1141        }1142    } else if (ctx->proj_type == PROJECTOR_TYPE_MERGER) {1143        embeddings = ggml_reshape_3d(ctx0, embeddings, hidden_size * 4, num_positions / 4, batch_size);1144 1145        embeddings = ggml_mul_mat(ctx0, model.mm_0_w, embeddings);1146        embeddings = ggml_add(ctx0, embeddings, model.mm_0_b);1147 1148        // GELU activation1149        embeddings = ggml_gelu(ctx0, embeddings);1150 1151        // Second linear layer1152        embeddings = ggml_mul_mat(ctx0, model.mm_1_w, embeddings);1153        embeddings = ggml_add(ctx0, embeddings, model.mm_1_b);1154    }1155 1156    // build the graph1157    ggml_build_forward_expand(gf, embeddings);1158 1159    ggml_free(ctx0);1160 1161    return gf;1162}1163 1164// read and create ggml_context containing the tensors and their data1165struct clip_ctx * clip_model_load(const char * fname, const int verbosity = 1) {1166    struct ggml_context * meta = NULL;1167 1168    struct gguf_init_params params = {1169        /*.no_alloc = */ true,1170        /*.ctx      = */ &meta,1171    };1172 1173    struct gguf_context * ctx = gguf_init_from_file(fname, params);1174    if (!ctx) {1175        throw std::runtime_error(format("%s: failed to load CLIP model from %s. Does this file exist?\n", __func__, fname));1176    }1177 1178    if (verbosity >= 1) {1179        const int n_tensors = gguf_get_n_tensors(ctx);1180        const int n_kv = gguf_get_n_kv(ctx);1181        const int ftype = get_u32(ctx, KEY_FTYPE);1182        const std::string ftype_str = get_ftype(ftype);1183        const int idx_desc = get_key_idx(ctx, KEY_DESCRIPTION);1184        const std::string description = gguf_get_val_str(ctx, idx_desc);1185        const int idx_name = gguf_find_key(ctx, KEY_NAME);1186        if (idx_name != -1) { // make name optional temporarily as some of the uploaded models missing it due to a bug1187            const std::string name = gguf_get_val_str(ctx, idx_name);1188            LOG_INF("%s: model name:   %s\n", __func__, name.c_str());1189        }1190        LOG_INF("%s: description:  %s\n", __func__, description.c_str());1191        LOG_INF("%s: GGUF version: %d\n", __func__, gguf_get_version(ctx));1192        LOG_INF("%s: alignment:    %zu\n", __func__, gguf_get_alignment(ctx));1193        LOG_INF("%s: n_tensors:    %d\n", __func__, n_tensors);1194        LOG_INF("%s: n_kv:         %d\n", __func__, n_kv);1195        LOG_INF("%s: ftype:        %s\n", __func__, ftype_str.c_str());1196        LOG_INF("\n");1197    }1198    const int n_tensors = gguf_get_n_tensors(ctx);1199 1200    // kv

Showing the first 1,200 of 2941 lines. Download the file for the rest.