KBaba7/llama.cpp
0
1// NOTE: This is modified from clip.cpp only for LLaVA,2// so there might be still unnecessary artifacts hanging around3// I'll gradually clean and extend it4// Note: Even when using identical normalized image inputs (see normalize_image_u8_to_f32()) we have a significant difference in resulting embeddings compared to pytorch5#include "clip.h"6#include "ggml.h"7#include "ggml-cpu.h"8#include "ggml-alloc.h"9#include "ggml-backend.h"10#include "gguf.h"11 12//#ifdef GGML_USE_CUDA13//#include "ggml-cuda.h"14//#endif15//16//#ifdef GGML_USE_SYCL17//#include "ggml-sycl.h"18//#endif19//20//#ifdef GGML_USE_METAL21//#include "ggml-metal.h"22//#endif23//24//#ifdef GGML_USE_CANN25//#include "ggml-cann.h"26//#endif27//28//#ifdef GGML_USE_VULKAN29//#include "ggml-vulkan.h"30//#endif31 32#define STB_IMAGE_IMPLEMENTATION33#include "stb_image.h"34 35#include <cassert>36#include <cmath>37#include <cstdlib>38#include <cstring>39#include <fstream>40#include <map>41#include <regex>42#include <stdexcept>43#include <vector>44#include <sstream>45#include <cinttypes>46#include <limits>47 48#if defined(LLAVA_LOG_OFF)49# define LOG_INF(...)50# define LOG_WRN(...)51# define LOG_ERR(...)52# define LOG_DBG(...)53#else // defined(LLAVA_LOG_OFF)54# define LOG_INF(...) do { fprintf(stdout, __VA_ARGS__); } while (0)55# define LOG_WRN(...) do { fprintf(stderr, __VA_ARGS__); } while (0)56# define LOG_ERR(...) do { fprintf(stderr, __VA_ARGS__); } while (0)57# define LOG_DBG(...) do { fprintf(stdout, __VA_ARGS__); } while (0)58#endif // defined(LLAVA_LOG_OFF)59 60//#define CLIP_DEBUG_FUNCTIONS61 62// RGB uint8 image63struct clip_image_u8 {64 int nx;65 int ny;66 67 std::vector<uint8_t> buf;68};69 70// RGB float32 image (NHWC)71// Memory layout: RGBRGBRGB...72struct clip_image_f32 {73 int nx;74 int ny;75 76 std::vector<float> buf;77};78 79static std::string format(const char * fmt, ...) {80 va_list ap;81 va_list ap2;82 va_start(ap, fmt);83 va_copy(ap2, ap);84 int size = vsnprintf(NULL, 0, fmt, ap);85 GGML_ASSERT(size >= 0 && size < INT_MAX); // NOLINT86 std::vector<char> buf(size + 1);87 int size2 = vsnprintf(buf.data(), size + 1, fmt, ap2);88 GGML_ASSERT(size2 == size);89 va_end(ap2);90 va_end(ap);91 return std::string(buf.data(), buf.size());92}93 94//95// key constants96//97 98#define KEY_FTYPE "general.file_type"99#define KEY_NAME "general.name"100#define KEY_DESCRIPTION "general.description"101#define KEY_HAS_TEXT_ENC "clip.has_text_encoder"102#define KEY_HAS_VIS_ENC "clip.has_vision_encoder"103#define KEY_HAS_LLAVA_PROJ "clip.has_llava_projector"104#define KEY_HAS_MINICPMV_PROJ "clip.has_minicpmv_projector"105#define KEY_HAS_GLM_PROJ "clip.has_glm_projector"106#define KEY_MINICPMV_VERSION "clip.minicpmv_version"107#define KEY_HAS_QWEN2VL_MERGER "clip.has_qwen2vl_merger"108#define KEY_USE_GELU "clip.use_gelu"109#define KEY_USE_SILU "clip.use_silu"110#define KEY_N_EMBD "clip.%s.embedding_length"111#define KEY_N_FF "clip.%s.feed_forward_length"112#define KEY_N_BLOCK "clip.%s.block_count"113#define KEY_N_HEAD "clip.%s.attention.head_count"114#define KEY_LAYER_NORM_EPS "clip.%s.attention.layer_norm_epsilon"115#define KEY_PROJ_DIM "clip.%s.projection_dim"116#define KEY_TOKENS "tokenizer.ggml.tokens"117#define KEY_N_POSITIONS "clip.text.context_length"118#define KEY_IMAGE_SIZE "clip.vision.image_size"119#define KEY_PATCH_SIZE "clip.vision.patch_size"120#define KEY_IMAGE_MEAN "clip.vision.image_mean"121#define KEY_IMAGE_STD "clip.vision.image_std"122#define KEY_PROJ_TYPE "clip.projector_type"123 124#define KEY_MM_PATCH_MERGE_TYPE "clip.vision.mm_patch_merge_type"125#define KEY_IMAGE_GRID_PINPOINTS "clip.vision.image_grid_pinpoints"126#define KEY_IMAGE_CROP_RESOLUTION "clip.vision.image_crop_resolution"127 128 129//130// tensor name constants131//132 133#define TN_TOKEN_EMBD "%s.token_embd.weight"134#define TN_POS_EMBD "%s.position_embd.weight"135#define TN_CLASS_EMBD "v.class_embd"136#define TN_PATCH_EMBD "v.patch_embd.weight" // not rename tensor with ".0" postfix for backwrad compat137#define TN_PATCH_EMBD_1 "v.patch_embd.weight.1"138#define TN_PATCH_BIAS "v.patch_embd.bias"139#define TN_ATTN_K "%s.blk.%d.attn_k.%s"140#define TN_ATTN_Q "%s.blk.%d.attn_q.%s"141#define TN_ATTN_V "%s.blk.%d.attn_v.%s"142#define TN_ATTN_OUTPUT "%s.blk.%d.attn_out.%s"143#define TN_FFN_DOWN "%s.blk.%d.ffn_down.%s"144#define TN_FFN_UP "%s.blk.%d.ffn_up.%s"145#define TN_LN_1 "%s.blk.%d.ln1.%s"146#define TN_LN_2 "%s.blk.%d.ln2.%s"147#define TN_LN_PRE "%s.pre_ln.%s"148#define TN_LN_POST "%s.post_ln.%s"149#define TN_TEXT_PROJ "text_projection.weight"150#define TN_VIS_PROJ "visual_projection.weight"151#define TN_LLAVA_PROJ "mm.%d.%s"152#define TN_MVLM_PROJ_MLP "mm.model.mlp.%d.%s"153#define TN_MVLM_PROJ_BLOCK "mm.model.mb_block.%d.block.%d.%s"154#define TN_MVLM_PROJ_PEG "mm.model.peg.%d.%s"155#define TN_IMAGE_NEWLINE "model.image_newline"156 157#define TN_MINICPMV_POS_EMBD_K "resampler.pos_embed_k"158#define TN_MINICPMV_QUERY "resampler.query"159#define TN_MINICPMV_PROJ "resampler.proj.weight"160#define TN_MINICPMV_KV_PROJ "resampler.kv.weight"161#define TN_MINICPMV_ATTN "resampler.attn.%s.%s"162#define TN_MINICPMV_LN "resampler.ln_%s.%s"163 164#define TN_GLM_ADAPER_CONV "adapter.conv.%s"165#define TN_GLM_ADAPTER_LINEAR "adapter.linear.linear.%s"166#define TN_GLM_ADAPTER_NORM_1 "adapter.linear.norm1.%s"167#define TN_GLM_ADAPTER_D_H_2_4H "adapter.linear.dense_h_to_4h.%s"168#define TN_GLM_ADAPTER_GATE "adapter.linear.gate.%s"169#define TN_GLM_ADAPTER_D_4H_2_H "adapter.linear.dense_4h_to_h.%s"170#define TN_GLM_BOI_W "adapter.boi"171#define TN_GLM_EOI_W "adapter.eoi"172 173 174enum projector_type {175 PROJECTOR_TYPE_MLP,176 PROJECTOR_TYPE_MLP_NORM,177 PROJECTOR_TYPE_LDP,178 PROJECTOR_TYPE_LDPV2,179 PROJECTOR_TYPE_RESAMPLER,180 PROJECTOR_TYPE_GLM_EDGE,181 PROJECTOR_TYPE_MERGER,182 PROJECTOR_TYPE_UNKNOWN,183};184 185static std::map<projector_type, std::string> PROJECTOR_TYPE_NAMES = {186 { PROJECTOR_TYPE_MLP, "mlp" },187 { PROJECTOR_TYPE_LDP, "ldp" },188 { PROJECTOR_TYPE_LDPV2, "ldpv2"},189 { PROJECTOR_TYPE_RESAMPLER, "resampler"},190 { PROJECTOR_TYPE_GLM_EDGE, "adapter"},191 { PROJECTOR_TYPE_MERGER, "qwen2vl_merger"},192};193 194 195//196// utilities to get data from a gguf file197//198 199static int get_key_idx(const gguf_context * ctx, const char * key) {200 int i = gguf_find_key(ctx, key);201 if (i == -1) {202 LOG_ERR("key %s not found in file\n", key);203 throw std::runtime_error(format("Missing required key: %s", key));204 }205 206 return i;207}208 209static uint32_t get_u32(const gguf_context * ctx, const std::string & key) {210 const int i = get_key_idx(ctx, key.c_str());211 212 return gguf_get_val_u32(ctx, i);213}214 215static float get_f32(const gguf_context * ctx, const std::string & key) {216 const int i = get_key_idx(ctx, key.c_str());217 218 return gguf_get_val_f32(ctx, i);219}220 221static struct ggml_tensor * get_tensor(struct ggml_context * ctx, const std::string & name) {222 struct ggml_tensor * cur = ggml_get_tensor(ctx, name.c_str());223 if (!cur) {224 throw std::runtime_error(format("%s: unable to find tensor %s\n", __func__, name.c_str()));225 }226 227 return cur;228}229 230static std::string get_ftype(int ftype) {231 return ggml_type_name(static_cast<ggml_type>(ftype));232}233 234static std::string gguf_data_to_str(enum gguf_type type, const void * data, int i) {235 switch (type) {236 case GGUF_TYPE_UINT8: return std::to_string(((const uint8_t *)data)[i]);237 case GGUF_TYPE_INT8: return std::to_string(((const int8_t *)data)[i]);238 case GGUF_TYPE_UINT16: return std::to_string(((const uint16_t *)data)[i]);239 case GGUF_TYPE_INT16: return std::to_string(((const int16_t *)data)[i]);240 case GGUF_TYPE_UINT32: return std::to_string(((const uint32_t *)data)[i]);241 case GGUF_TYPE_INT32: return std::to_string(((const int32_t *)data)[i]);242 case GGUF_TYPE_UINT64: return std::to_string(((const uint64_t *)data)[i]);243 case GGUF_TYPE_INT64: return std::to_string(((const int64_t *)data)[i]);244 case GGUF_TYPE_FLOAT32: return std::to_string(((const float *)data)[i]);245 case GGUF_TYPE_FLOAT64: return std::to_string(((const double *)data)[i]);246 case GGUF_TYPE_BOOL: return ((const bool *)data)[i] ? "true" : "false";247 default: return format("unknown type %d", type);248 }249}250 251static void replace_all(std::string & s, const std::string & search, const std::string & replace) {252 if (search.empty()) {253 return;254 }255 std::string builder;256 builder.reserve(s.length());257 size_t pos = 0;258 size_t last_pos = 0;259 while ((pos = s.find(search, last_pos)) != std::string::npos) {260 builder.append(s, last_pos, pos - last_pos);261 builder.append(replace);262 last_pos = pos + search.length();263 }264 builder.append(s, last_pos, std::string::npos);265 s = std::move(builder);266}267 268static std::string gguf_kv_to_str(const struct gguf_context * ctx_gguf, int i) {269 const enum gguf_type type = gguf_get_kv_type(ctx_gguf, i);270 271 switch (type) {272 case GGUF_TYPE_STRING:273 return gguf_get_val_str(ctx_gguf, i);274 case GGUF_TYPE_ARRAY:275 {276 const enum gguf_type arr_type = gguf_get_arr_type(ctx_gguf, i);277 int arr_n = gguf_get_arr_n(ctx_gguf, i);278 const void * data = arr_type == GGUF_TYPE_STRING ? nullptr : gguf_get_arr_data(ctx_gguf, i);279 std::stringstream ss;280 ss << "[";281 for (int j = 0; j < arr_n; j++) {282 if (arr_type == GGUF_TYPE_STRING) {283 std::string val = gguf_get_arr_str(ctx_gguf, i, j);284 // escape quotes285 replace_all(val, "\\", "\\\\");286 replace_all(val, "\"", "\\\"");287 ss << '"' << val << '"';288 } else if (arr_type == GGUF_TYPE_ARRAY) {289 ss << "???";290 } else {291 ss << gguf_data_to_str(arr_type, data, j);292 }293 if (j < arr_n - 1) {294 ss << ", ";295 }296 }297 ss << "]";298 return ss.str();299 }300 default:301 return gguf_data_to_str(type, gguf_get_val_data(ctx_gguf, i), 0);302 }303}304 305static void print_tensor_info(const ggml_tensor * tensor, const char * prefix = "") {306 size_t tensor_size = ggml_nbytes(tensor);307 LOG_INF("%s: n_dims = %d, name = %s, tensor_size=%zu, shape:[%" PRId64 ", %" PRId64 ", %" PRId64 ", %" PRId64 "], type = %s\n",308 prefix, ggml_n_dims(tensor), tensor->name, tensor_size,309 tensor->ne[0], tensor->ne[1], tensor->ne[2], tensor->ne[3], ggml_type_name(tensor->type));310}311 312static projector_type clip_projector_type_from_string(const std::string & name) {313 for (const auto & kv : PROJECTOR_TYPE_NAMES) { // NOLINT314 if (kv.second == name) {315 return kv.first;316 }317 }318 return PROJECTOR_TYPE_UNKNOWN;319}320 321#ifdef CLIP_DEBUG_FUNCTIONS322static void clip_image_write_image_to_ppm(const clip_image_u8& img, const std::string& filename) {323 std::ofstream file(filename, std::ios::binary);324 if (!file.is_open()) {325 LOG_ERR("Failed to open file for writing: %s\n", filename.c_str());326 return;327 }328 329 // PPM header: P6 format, width, height, and max color value330 file << "P6\n" << img.nx << " " << img.ny << "\n255\n";331 332 // Write pixel data333 for (size_t i = 0; i < img.buf.size(); i += 3) {334 // PPM expects binary data in RGB format, which matches our image buffer335 file.write(reinterpret_cast<const char*>(&img.buf[i]), 3);336 }337 338 file.close();339}340 341static void clip_image_save_to_bmp(const clip_image_u8& img, const std::string& filename) {342 std::ofstream file(filename, std::ios::binary);343 if (!file.is_open()) {344 LOG_ERR("Failed to open file for writing: %s\n", filename.c_str());345 return;346 }347 348 int fileSize = 54 + 3 * img.nx * img.ny; // File header + info header + pixel data349 int bytesPerPixel = 3;350 int widthInBytes = img.nx * bytesPerPixel;351 int paddingAmount = (4 - (widthInBytes % 4)) % 4;352 int stride = widthInBytes + paddingAmount;353 354 // Bitmap file header355 unsigned char fileHeader[14] = {356 'B','M', // Signature357 0,0,0,0, // Image file size in bytes358 0,0,0,0, // Reserved359 54,0,0,0 // Start of pixel array360 };361 362 // Total file size363 fileSize = 54 + (stride * img.ny);364 fileHeader[2] = (unsigned char)(fileSize);365 fileHeader[3] = (unsigned char)(fileSize >> 8);366 fileHeader[4] = (unsigned char)(fileSize >> 16);367 fileHeader[5] = (unsigned char)(fileSize >> 24);368 369 // Bitmap information header (BITMAPINFOHEADER)370 unsigned char infoHeader[40] = {371 40,0,0,0, // Size of this header (40 bytes)372 0,0,0,0, // Image width373 0,0,0,0, // Image height374 1,0, // Number of color planes375 24,0, // Bits per pixel376 0,0,0,0, // No compression377 0,0,0,0, // Image size (can be 0 for no compression)378 0,0,0,0, // X pixels per meter (not specified)379 0,0,0,0, // Y pixels per meter (not specified)380 0,0,0,0, // Total colors (color table not used)381 0,0,0,0 // Important colors (all are important)382 };383 384 // Width and height in the information header385 infoHeader[4] = (unsigned char)(img.nx);386 infoHeader[5] = (unsigned char)(img.nx >> 8);387 infoHeader[6] = (unsigned char)(img.nx >> 16);388 infoHeader[7] = (unsigned char)(img.nx >> 24);389 infoHeader[8] = (unsigned char)(img.ny);390 infoHeader[9] = (unsigned char)(img.ny >> 8);391 infoHeader[10] = (unsigned char)(img.ny >> 16);392 infoHeader[11] = (unsigned char)(img.ny >> 24);393 394 // Write file headers395 file.write(reinterpret_cast<char*>(fileHeader), sizeof(fileHeader));396 file.write(reinterpret_cast<char*>(infoHeader), sizeof(infoHeader));397 398 // Pixel data399 std::vector<unsigned char> padding(3, 0); // Max padding size to be added to each row400 for (int y = img.ny - 1; y >= 0; --y) { // BMP files are stored bottom-to-top401 for (int x = 0; x < img.nx; ++x) {402 // Each pixel403 size_t pixelIndex = (y * img.nx + x) * 3;404 unsigned char pixel[3] = {405 img.buf[pixelIndex + 2], // BMP stores pixels in BGR format406 img.buf[pixelIndex + 1],407 img.buf[pixelIndex]408 };409 file.write(reinterpret_cast<char*>(pixel), 3);410 }411 // Write padding for the row412 file.write(reinterpret_cast<char*>(padding.data()), paddingAmount);413 }414 415 file.close();416}417 418// debug function to convert f32 to u8419static void clip_image_convert_f32_to_u8(const clip_image_f32& src, clip_image_u8& dst) {420 dst.nx = src.nx;421 dst.ny = src.ny;422 dst.buf.resize(3 * src.nx * src.ny);423 for (size_t i = 0; i < src.buf.size(); ++i) {424 dst.buf[i] = static_cast<uint8_t>(std::min(std::max(int(src.buf[i] * 255.0f), 0), 255));425 }426}427#endif428 429 430//431// clip layers432//433 434struct clip_hparams {435 int32_t image_size;436 int32_t patch_size;437 int32_t hidden_size;438 int32_t n_intermediate;439 int32_t projection_dim;440 int32_t n_head;441 int32_t n_layer;442 443 float eps;444 445 char mm_patch_merge_type[32] = "flat"; // spatial_unpad or flat (default)446 447 int32_t image_grid_pinpoints[32];448 int32_t image_crop_resolution;449};450 451struct clip_layer {452 // attention453 struct ggml_tensor * k_w;454 struct ggml_tensor * k_b;455 struct ggml_tensor * q_w;456 struct ggml_tensor * q_b;457 struct ggml_tensor * v_w;458 struct ggml_tensor * v_b;459 460 struct ggml_tensor * o_w;461 struct ggml_tensor * o_b;462 463 // layernorm 1464 struct ggml_tensor * ln_1_w;465 struct ggml_tensor * ln_1_b;466 467 // ff468 struct ggml_tensor * ff_i_w;469 struct ggml_tensor * ff_i_b;470 471 struct ggml_tensor * ff_o_w;472 struct ggml_tensor * ff_o_b;473 474 // layernorm 2475 struct ggml_tensor * ln_2_w;476 struct ggml_tensor * ln_2_b;477};478 479struct clip_vision_model {480 struct clip_hparams hparams;481 482 // embeddings483 struct ggml_tensor * class_embedding;484 struct ggml_tensor * patch_embeddings_0;485 struct ggml_tensor * patch_embeddings_1; // second Conv2D kernel when we decouple Conv3D along temproal dimension (Qwen2VL)486 struct ggml_tensor * patch_bias;487 struct ggml_tensor * position_embeddings;488 489 struct ggml_tensor * pre_ln_w;490 struct ggml_tensor * pre_ln_b;491 492 std::vector<clip_layer> layers;493 494 struct ggml_tensor * post_ln_w;495 struct ggml_tensor * post_ln_b;496 497 struct ggml_tensor * projection;498 499 // LLaVA projection500 struct ggml_tensor * mm_0_w = NULL;501 struct ggml_tensor * mm_0_b = NULL;502 struct ggml_tensor * mm_2_w = NULL;503 struct ggml_tensor * mm_2_b = NULL;504 505 struct ggml_tensor * image_newline = NULL;506 507 // Yi type models with mlp+normalization projection508 struct ggml_tensor * mm_1_w = NULL; // Yi type models have 0, 1, 3, 4509 struct ggml_tensor * mm_1_b = NULL;510 struct ggml_tensor * mm_3_w = NULL;511 struct ggml_tensor * mm_3_b = NULL;512 struct ggml_tensor * mm_4_w = NULL;513 struct ggml_tensor * mm_4_b = NULL;514 515 //GLMV-Edge projection516 struct ggml_tensor * mm_model_adapter_conv_w;517 struct ggml_tensor * mm_model_adapter_conv_b;518 struct ggml_tensor * boi_w;519 struct ggml_tensor * eoi_w;520 521 // MobileVLM projection522 struct ggml_tensor * mm_model_mlp_1_w;523 struct ggml_tensor * mm_model_mlp_1_b;524 struct ggml_tensor * mm_model_mlp_3_w;525 struct ggml_tensor * mm_model_mlp_3_b;526 struct ggml_tensor * mm_model_block_1_block_0_0_w;527 struct ggml_tensor * mm_model_block_1_block_0_1_w;528 struct ggml_tensor * mm_model_block_1_block_0_1_b;529 struct ggml_tensor * mm_model_block_1_block_1_fc1_w;530 struct ggml_tensor * mm_model_block_1_block_1_fc1_b;531 struct ggml_tensor * mm_model_block_1_block_1_fc2_w;532 struct ggml_tensor * mm_model_block_1_block_1_fc2_b;533 struct ggml_tensor * mm_model_block_1_block_2_0_w;534 struct ggml_tensor * mm_model_block_1_block_2_1_w;535 struct ggml_tensor * mm_model_block_1_block_2_1_b;536 struct ggml_tensor * mm_model_block_2_block_0_0_w;537 struct ggml_tensor * mm_model_block_2_block_0_1_w;538 struct ggml_tensor * mm_model_block_2_block_0_1_b;539 struct ggml_tensor * mm_model_block_2_block_1_fc1_w;540 struct ggml_tensor * mm_model_block_2_block_1_fc1_b;541 struct ggml_tensor * mm_model_block_2_block_1_fc2_w;542 struct ggml_tensor * mm_model_block_2_block_1_fc2_b;543 struct ggml_tensor * mm_model_block_2_block_2_0_w;544 struct ggml_tensor * mm_model_block_2_block_2_1_w;545 struct ggml_tensor * mm_model_block_2_block_2_1_b;546 547 // MobileVLM_V2 projection548 struct ggml_tensor * mm_model_mlp_0_w;549 struct ggml_tensor * mm_model_mlp_0_b;550 struct ggml_tensor * mm_model_mlp_2_w;551 struct ggml_tensor * mm_model_mlp_2_b;552 struct ggml_tensor * mm_model_peg_0_w;553 struct ggml_tensor * mm_model_peg_0_b;554 555 // MINICPMV projection556 struct ggml_tensor * mm_model_pos_embed_k;557 struct ggml_tensor * mm_model_query;558 struct ggml_tensor * mm_model_proj;559 struct ggml_tensor * mm_model_kv_proj;560 struct ggml_tensor * mm_model_attn_q_w;561 struct ggml_tensor * mm_model_attn_q_b;562 struct ggml_tensor * mm_model_attn_k_w;563 struct ggml_tensor * mm_model_attn_k_b;564 struct ggml_tensor * mm_model_attn_v_w;565 struct ggml_tensor * mm_model_attn_v_b;566 struct ggml_tensor * mm_model_attn_o_w;567 struct ggml_tensor * mm_model_attn_o_b;568 struct ggml_tensor * mm_model_ln_q_w;569 struct ggml_tensor * mm_model_ln_q_b;570 struct ggml_tensor * mm_model_ln_kv_w;571 struct ggml_tensor * mm_model_ln_kv_b;572 struct ggml_tensor * mm_model_ln_post_w;573 struct ggml_tensor * mm_model_ln_post_b;574};575 576struct clip_ctx {577 bool has_text_encoder = false;578 bool has_vision_encoder = false;579 bool has_llava_projector = false;580 bool has_minicpmv_projector = false;581 bool has_glm_projector = false;582 bool has_qwen2vl_merger = false;583 int minicpmv_version = 2;584 585 struct clip_vision_model vision_model;586 projector_type proj_type = PROJECTOR_TYPE_MLP;587 588 float image_mean[3];589 float image_std[3];590 bool use_gelu = false;591 bool use_silu = false;592 int32_t ftype = 1;593 594 bool has_class_embedding = true;595 bool has_pre_norm = true;596 bool has_post_norm = false;597 bool has_patch_bias = false;598 599 struct gguf_context * ctx_gguf;600 struct ggml_context * ctx_data;601 602 std::vector<uint8_t> buf_compute_meta;603 604 // memory buffers to evaluate the model605 ggml_backend_buffer_t params_buffer = NULL;606 607 ggml_backend_t backend = NULL;608 ggml_gallocr_t compute_alloc = NULL;609 610 struct clip_image_size * load_image_size;611};612 613static ggml_cgraph * clip_image_build_graph(clip_ctx * ctx, const clip_image_f32_batch * imgs, struct clip_image_size * load_image_size, bool is_inf = false) {614 if (!ctx->has_vision_encoder) {615 LOG_ERR("This gguf file seems to have no vision encoder\n");616 return nullptr;617 }618 619 const auto & model = ctx->vision_model;620 const auto & hparams = model.hparams;621 622 const int image_size = hparams.image_size;623 int image_size_width = image_size;624 int image_size_height = image_size;625 if (ctx->has_minicpmv_projector) {626 if (load_image_size == nullptr) {627 load_image_size = clip_image_size_init();628 }629 LOG_DBG("%s: %d %d\n", __func__, load_image_size->width, load_image_size->height);630 image_size_width = load_image_size->width;631 image_size_height = load_image_size->height;632 if (is_inf) {633 image_size_width = imgs->data->nx;634 image_size_height = imgs->data->ny;635 }636 }637 else if (ctx->has_qwen2vl_merger) {638 // use the image's native resolution when image is avaible639 if (is_inf) {640 // if (imgs->data->nx && imgs->data->ny) {641 image_size_width = imgs->data->nx;642 image_size_height = imgs->data->ny;643 }644 }645 const int patch_size = hparams.patch_size;646 const int num_patches = ((image_size_width / patch_size) * (image_size_height / patch_size));647 const int patches_w = image_size_width / patch_size;648 const int patches_h = image_size_height / patch_size;649 const int num_positions = num_patches + (ctx->has_class_embedding ? 1 : 0);650 const int num_position_ids = ctx->has_qwen2vl_merger ? num_positions * 4 : num_positions;651 const int hidden_size = hparams.hidden_size;652 const int n_head = hparams.n_head;653 const int d_head = hidden_size / n_head;654 int n_layer = hparams.n_layer;655 const float eps = hparams.eps;656 int mrope_sections[4] = {d_head/4, d_head/4, d_head/4, d_head/4};657 658 const int batch_size = imgs->size;659 660 if (ctx->has_llava_projector || ctx->has_minicpmv_projector || ctx->has_glm_projector) {661 GGML_ASSERT(batch_size == 1);662 }663 664 struct ggml_init_params params = {665 /*.mem_size =*/ ctx->buf_compute_meta.size(),666 /*.mem_buffer =*/ ctx->buf_compute_meta.data(),667 /*.no_alloc =*/ true,668 };669 670 struct ggml_context * ctx0 = ggml_init(params);671 struct ggml_cgraph * gf = ggml_new_graph(ctx0);672 673 struct ggml_tensor * inp_raw = ggml_new_tensor_4d(ctx0, GGML_TYPE_F32, image_size_width, image_size_height, 3, batch_size);674 ggml_set_name(inp_raw, "inp_raw");675 ggml_set_input(inp_raw);676 677 struct ggml_tensor * inp = ggml_conv_2d(ctx0, model.patch_embeddings_0, inp_raw, patch_size, patch_size, 0, 0, 1, 1);678 679 if (ctx->has_qwen2vl_merger) {680 GGML_ASSERT(image_size_width % (patch_size * 2) == 0);681 GGML_ASSERT(image_size_height % (patch_size * 2) == 0);682 683 auto inp_1 = ggml_conv_2d(ctx0, model.patch_embeddings_1, inp_raw, patch_size, patch_size, 0, 0, 1, 1);684 inp = ggml_add(ctx0, inp, inp_1);685 inp = ggml_cont(ctx0, ggml_permute(ctx0, inp, 1, 2, 0, 3)); // [w, h, c, b] -> [c, w, h, b]686 inp = ggml_reshape_4d(687 ctx0, inp,688 hidden_size * 2, patches_w / 2, patches_h, batch_size);689 inp = ggml_reshape_4d(690 ctx0, inp,691 hidden_size * 2, patches_w / 2, 2, batch_size * (patches_h / 2));692 inp = ggml_cont(ctx0, ggml_permute(ctx0, inp, 0, 2, 1, 3));693 inp = ggml_reshape_3d(694 ctx0, inp,695 hidden_size, patches_w * patches_h, batch_size);696 }697 else {698 inp = ggml_reshape_3d(ctx0, inp, num_patches, hidden_size, batch_size);699 inp = ggml_cont(ctx0, ggml_permute(ctx0, inp, 1, 0, 2, 3));700 }701 702 if (ctx->has_patch_bias) {703 // inp = ggml_add(ctx0, inp, ggml_repeat(ctx0, model.patch_bias, inp));704 inp = ggml_add(ctx0, inp, model.patch_bias);705 }706 struct ggml_tensor * embeddings = inp;707 struct ggml_tensor * pos_embed = nullptr;708 709 if (ctx->has_llava_projector) {710 // concat class_embeddings and patch_embeddings711 if (ctx->has_class_embedding) {712 embeddings = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, hidden_size, num_positions, batch_size);713 ggml_set_name(embeddings, "embeddings");714 ggml_set_input(embeddings);715 embeddings = ggml_acc(ctx0, embeddings, model.class_embedding,716 embeddings->nb[1], embeddings->nb[2], embeddings->nb[3], 0);717 embeddings = ggml_acc(ctx0, embeddings, inp,718 embeddings->nb[1], embeddings->nb[2], embeddings->nb[3], model.class_embedding->nb[1]);719 }720 }721 722 struct ggml_tensor * positions = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, num_position_ids);723 ggml_set_name(positions, "positions");724 ggml_set_input(positions);725 726 if (!ctx->has_qwen2vl_merger) { // qwen2vl use rope position embedding727 embeddings =728 ggml_add(ctx0, embeddings, ggml_get_rows(ctx0, model.position_embeddings, positions));729 }730 731 if (ctx->has_minicpmv_projector) {732 int pos_w = image_size_width/patch_size;733 int pos_h = image_size_height/patch_size;734 if (ctx->minicpmv_version == 2) {735 pos_embed = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 4096, pos_w * pos_h, 1);736 }737 else if (ctx->minicpmv_version == 3) {738 pos_embed = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 3584, pos_w * pos_h, 1);739 }740 else if (ctx->minicpmv_version == 4) {741 pos_embed = ggml_new_tensor_3d(ctx0, GGML_TYPE_F32, 3584, pos_w * pos_h, 1);742 }743 ggml_set_name(pos_embed, "pos_embed");744 ggml_set_input(pos_embed);745 }746 747 // pre-layernorm748 if (ctx->has_pre_norm) {749 embeddings = ggml_norm(ctx0, embeddings, eps);750 ggml_set_name(embeddings, "pre_ln");751 752 embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.pre_ln_w), model.pre_ln_b);753 }754 755 // loop over layers756 if (ctx->has_minicpmv_projector || ctx->has_glm_projector || ctx->has_qwen2vl_merger) {757 n_layer += 1;758 }759 for (int il = 0; il < n_layer - 1; il++) {760 struct ggml_tensor * cur = embeddings; // embeddings = residual, cur = hidden_states761 762 //const size_t nb_q_w = model.layers[il].q_w->nb[0];763 764 // layernorm1765 {766 cur = ggml_norm(ctx0, cur, eps);767 768 cur = ggml_add(ctx0, ggml_mul(ctx0, cur, model.layers[il].ln_1_w),769 model.layers[il].ln_1_b);770 }771 772 // self-attention773 {774 775 struct ggml_tensor * Q =776 ggml_add(ctx0, ggml_mul_mat(ctx0, model.layers[il].q_w, cur), model.layers[il].q_b);777 778 Q = ggml_reshape_4d(ctx0, Q, d_head, n_head, num_positions, batch_size);779 if (ctx->has_qwen2vl_merger) {780 Q = ggml_rope_multi(781 ctx0, Q, positions, nullptr,782 d_head/2, mrope_sections, GGML_ROPE_TYPE_VISION, 32768, 10000, 1, 0, 1, 32, 1);783 }784 Q = ggml_scale_inplace(ctx0, Q, 1.0f / sqrt((float)d_head));785 Q = ggml_cont(ctx0, ggml_permute(ctx0, Q, 0, 2, 1, 3));786 Q = ggml_reshape_3d(ctx0, Q, d_head, num_positions, n_head * batch_size);787 788 struct ggml_tensor * K =789 ggml_add(ctx0, ggml_mul_mat(ctx0, model.layers[il].k_w, cur), model.layers[il].k_b);790 791 K = ggml_reshape_4d(ctx0, K, d_head, n_head, num_positions, batch_size);792 if (ctx->has_qwen2vl_merger) {793 K = ggml_rope_multi(794 ctx0, K, positions, nullptr,795 d_head/2, mrope_sections, GGML_ROPE_TYPE_VISION, 32768, 10000, 1, 0, 1, 32, 1);796 }797 K = ggml_cont(ctx0, ggml_permute(ctx0, K, 0, 2, 1, 3));798 K = ggml_reshape_3d(ctx0, K, d_head, num_positions, n_head * batch_size);799 800 struct ggml_tensor * V =801 ggml_add(ctx0, ggml_mul_mat(ctx0, model.layers[il].v_w, cur), model.layers[il].v_b);802 803 V = ggml_reshape_4d(ctx0, V, d_head, n_head, num_positions, batch_size);804 V = ggml_cont(ctx0, ggml_permute(ctx0, V, 1, 2, 0, 3));805 V = ggml_reshape_3d(ctx0, V, num_positions, d_head, n_head * batch_size);806 807 struct ggml_tensor * KQ = ggml_mul_mat(ctx0, K, Q);808 KQ = ggml_soft_max_inplace(ctx0, KQ);809 struct ggml_tensor * KQV = ggml_mul_mat(ctx0, V, KQ);810 KQV = ggml_reshape_4d(ctx0, KQV, d_head, num_positions, n_head, batch_size);811 KQV = ggml_permute(ctx0, KQV, 0, 2, 1, 3);812 813 cur = ggml_cont_3d(ctx0, KQV, hidden_size, num_positions, batch_size);814 }815 816 // attention output817 cur = ggml_add(ctx0, ggml_mul_mat(ctx0, model.layers[il].o_w, cur), model.layers[il].o_b);818 819 // re-add the layer input, e.g., residual820 cur = ggml_add(ctx0, cur, embeddings);821 822 embeddings = cur; // embeddings = residual, cur = hidden_states823 824 // layernorm2825 {826 cur = ggml_norm(ctx0, cur, eps);827 828 cur = ggml_add(ctx0, ggml_mul(ctx0, cur, model.layers[il].ln_2_w), model.layers[il].ln_2_b);829 }830 831 cur = ggml_mul_mat(ctx0, model.layers[il].ff_i_w, cur);832 cur = ggml_add(ctx0, cur, model.layers[il].ff_i_b);833 834 if (ctx->use_gelu) {835 cur = ggml_gelu_inplace(ctx0, cur);836 } else if (ctx->use_silu) {837 cur = ggml_silu_inplace(ctx0, cur);838 } else {839 cur = ggml_gelu_quick_inplace(ctx0, cur);840 }841 842 cur = ggml_mul_mat(ctx0, model.layers[il].ff_o_w, cur);843 cur = ggml_add(ctx0, cur, model.layers[il].ff_o_b);844 845 // residual 2846 cur = ggml_add(ctx0, embeddings, cur);847 848 embeddings = cur;849 850 }851 852 // post-layernorm853 if (ctx->has_post_norm) {854 embeddings = ggml_norm(ctx0, embeddings, eps);855 ggml_set_name(embeddings, "post_ln");856 857 embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.post_ln_w), model.post_ln_b);858 }859 860 // llava projector861 if (ctx->has_llava_projector) {862 embeddings = ggml_reshape_2d(ctx0, embeddings, embeddings->ne[0], embeddings->ne[1]);863 864 struct ggml_tensor * patches = ggml_new_tensor_1d(ctx0, GGML_TYPE_I32, num_patches);865 ggml_set_name(patches, "patches");866 ggml_set_input(patches);867 868 // shape [1, 576, 1024]869 // ne is whcn, ne = [1024, 576, 1, 1]870 embeddings = ggml_get_rows(ctx0, embeddings, patches);871 872 // print_tensor_info(embeddings, "embeddings");873 874 // llava projector875 if (ctx->proj_type == PROJECTOR_TYPE_MLP) {876 embeddings = ggml_mul_mat(ctx0, model.mm_0_w, embeddings);877 embeddings = ggml_add(ctx0, embeddings, model.mm_0_b);878 879 embeddings = ggml_gelu(ctx0, embeddings);880 embeddings = ggml_mul_mat(ctx0, model.mm_2_w, embeddings);881 embeddings = ggml_add(ctx0, embeddings, model.mm_2_b);882 }883 else if (ctx->proj_type == PROJECTOR_TYPE_MLP_NORM) {884 embeddings = ggml_mul_mat(ctx0, model.mm_0_w, embeddings);885 embeddings = ggml_add(ctx0, embeddings, model.mm_0_b);886 // ggml_tensor_printf(embeddings, "mm_0_w",0,true,false);887 // First LayerNorm888 embeddings = ggml_norm(ctx0, embeddings, eps);889 embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_1_w),890 model.mm_1_b);891 892 // GELU activation893 embeddings = ggml_gelu(ctx0, embeddings);894 895 // Second linear layer896 embeddings = ggml_mul_mat(ctx0, model.mm_3_w, embeddings);897 embeddings = ggml_add(ctx0, embeddings, model.mm_3_b);898 899 // Second LayerNorm900 embeddings = ggml_norm(ctx0, embeddings, eps);901 embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_4_w),902 model.mm_4_b);903 }904 else if (ctx->proj_type == PROJECTOR_TYPE_LDP) {905 // MobileVLM projector906 int n_patch = 24;907 struct ggml_tensor * mlp_1 = ggml_mul_mat(ctx0, model.mm_model_mlp_1_w, embeddings);908 mlp_1 = ggml_add(ctx0, mlp_1, model.mm_model_mlp_1_b);909 mlp_1 = ggml_gelu(ctx0, mlp_1);910 struct ggml_tensor * mlp_3 = ggml_mul_mat(ctx0, model.mm_model_mlp_3_w, mlp_1);911 mlp_3 = ggml_add(ctx0, mlp_3, model.mm_model_mlp_3_b);912 // mlp_3 shape = [1, 576, 2048], ne = [2048, 576, 1, 1]913 914 // block 1915 struct ggml_tensor * block_1 = nullptr;916 {917 // transpose from [1, 576, 2048] --> [1, 2048, 576] --> [1, 2048, 24, 24]918 mlp_3 = ggml_cont(ctx0, ggml_permute(ctx0, mlp_3, 1, 0, 2, 3));919 mlp_3 = ggml_reshape_4d(ctx0, mlp_3, n_patch, n_patch, mlp_3->ne[1], mlp_3->ne[2]);920 // stride = 1, padding = 1, bias is nullptr921 block_1 = ggml_conv_2d_dw(ctx0, model.mm_model_block_1_block_0_0_w, mlp_3, 1, 1, 1, 1, 1, 1);922 923 // layer norm924 // // block_1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]925 block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 2, 0, 3));926 // block_1 shape = [1, 24, 24, 2048], ne = [2048, 24, 24, 1]927 block_1 = ggml_norm(ctx0, block_1, eps);928 block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_1_block_0_1_w), model.mm_model_block_1_block_0_1_b);929 block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));930 931 // block_1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]932 // hardswish933 struct ggml_tensor * block_1_hw = ggml_hardswish(ctx0, block_1);934 935 block_1 = ggml_pool_2d(ctx0, block_1_hw, GGML_OP_POOL_AVG, block_1_hw->ne[0], block_1_hw->ne[1], block_1_hw->ne[0], block_1_hw->ne[1], 0, 0);936 // block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]937 // pointwise conv938 block_1 = ggml_reshape_2d(ctx0, block_1, block_1->ne[0]*block_1->ne[1]*block_1->ne[2], block_1->ne[3]);939 block_1 = ggml_mul_mat(ctx0, model.mm_model_block_1_block_1_fc1_w, block_1);940 block_1 = ggml_add(ctx0, block_1, model.mm_model_block_1_block_1_fc1_b);941 block_1 = ggml_relu(ctx0, block_1);942 block_1 = ggml_mul_mat(ctx0, model.mm_model_block_1_block_1_fc2_w, block_1);943 block_1 = ggml_add(ctx0, block_1, model.mm_model_block_1_block_1_fc2_b);944 block_1 = ggml_hardsigmoid(ctx0, block_1);945 // block_1_hw shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1], block_1 shape = [1, 2048], ne = [2048, 1, 1, 1]946 block_1 = ggml_reshape_4d(ctx0, block_1, 1, 1, block_1->ne[0], block_1->ne[1]);947 block_1 = ggml_mul(ctx0, block_1_hw, block_1);948 949 int w = block_1->ne[0], h = block_1->ne[1];950 block_1 = ggml_reshape_3d(ctx0, block_1, w*h, block_1->ne[2], block_1->ne[3]);951 block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 0, 2, 3));952 953 // block_1 shape = [1, 24*24, 2048], ne = [24*24, 2048, 1]954 block_1 = ggml_mul_mat(ctx0, model.mm_model_block_1_block_2_0_w, block_1);955 block_1 = ggml_reshape_4d(ctx0, block_1, block_1->ne[0], w, h, block_1->ne[3]);956 957 // block_1 shape = [1, 24, 24, 2048], ne = [2048, 24, 24, 1]958 block_1 = ggml_norm(ctx0, block_1, eps);959 block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_1_block_2_1_w), model.mm_model_block_1_block_2_1_b);960 block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));961 // block1 shape = [1, 2048, 24, 24], ne = [24, 24, 2048, 1]962 // residual963 block_1 = ggml_add(ctx0, mlp_3, block_1);964 }965 966 // block_2967 {968 // stride = 2969 block_1 = ggml_conv_2d_dw(ctx0, model.mm_model_block_2_block_0_0_w, block_1, 2, 2, 1, 1, 1, 1);970 971 // block_1 shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1]972 // layer norm973 block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 2, 0, 3));974 // block_1 shape = [1, 12, 12, 2048], ne = [2048, 12, 12, 1]975 block_1 = ggml_norm(ctx0, block_1, eps);976 block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_2_block_0_1_w), model.mm_model_block_2_block_0_1_b);977 block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 2, 0, 1, 3));978 // block_1 shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1]979 // hardswish980 struct ggml_tensor * block_1_hw = ggml_hardswish(ctx0, block_1);981 982 // not sure the parameters is right for globalAvgPooling983 block_1 = ggml_pool_2d(ctx0, block_1_hw, GGML_OP_POOL_AVG, block_1_hw->ne[0], block_1_hw->ne[1], block_1_hw->ne[0], block_1_hw->ne[1], 0, 0);984 // block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]985 // pointwise conv986 block_1 = ggml_reshape_2d(ctx0, block_1, block_1->ne[0]*block_1->ne[1]*block_1->ne[2], block_1->ne[3]);987 block_1 = ggml_mul_mat(ctx0, model.mm_model_block_2_block_1_fc1_w, block_1);988 block_1 = ggml_add(ctx0, block_1, model.mm_model_block_2_block_1_fc1_b);989 block_1 = ggml_relu(ctx0, block_1);990 block_1 = ggml_mul_mat(ctx0, model.mm_model_block_2_block_1_fc2_w, block_1);991 block_1 = ggml_add(ctx0, block_1, model.mm_model_block_2_block_1_fc2_b);992 block_1 = ggml_hardsigmoid(ctx0, block_1);993 994 // block_1_hw shape = [1, 2048, 12, 12], ne = [12, 12, 2048, 1], block_1 shape = [1, 2048, 1, 1], ne = [1, 1, 2048, 1]995 block_1 = ggml_reshape_4d(ctx0, block_1, 1, 1, block_1->ne[0], block_1->ne[1]);996 block_1 = ggml_mul(ctx0, block_1_hw, block_1);997 998 int w = block_1->ne[0], h = block_1->ne[1];999 block_1 = ggml_reshape_3d(ctx0, block_1, w*h, block_1->ne[2], block_1->ne[3]);1000 block_1 = ggml_cont(ctx0, ggml_permute(ctx0, block_1, 1, 0, 2, 3));1001 // block_1 shape = [1, 24*24, 2048], ne = [24*24, 2048, 1]1002 block_1 = ggml_mul_mat(ctx0, model.mm_model_block_2_block_2_0_w, block_1);1003 block_1 = ggml_reshape_4d(ctx0, block_1, block_1->ne[0], w, h, block_1->ne[3]);1004 1005 1006 // block_1 shape = [1, 12, 12, 2048], ne = [2048, 12, 12, 1]1007 block_1 = ggml_norm(ctx0, block_1, eps);1008 block_1 = ggml_add(ctx0, ggml_mul(ctx0, block_1, model.mm_model_block_2_block_2_1_w), model.mm_model_block_2_block_2_1_b);1009 block_1 = ggml_reshape_3d(ctx0, block_1, block_1->ne[0], block_1->ne[1] * block_1->ne[2], block_1->ne[3]);1010 // block_1 shape = [1, 144, 2048], ne = [2048, 144, 1]1011 }1012 embeddings = block_1;1013 }1014 else if (ctx->proj_type == PROJECTOR_TYPE_LDPV2)1015 {1016 int n_patch = 24;1017 struct ggml_tensor * mlp_0 = ggml_mul_mat(ctx0, model.mm_model_mlp_0_w, embeddings);1018 mlp_0 = ggml_add(ctx0, mlp_0, model.mm_model_mlp_0_b);1019 mlp_0 = ggml_gelu(ctx0, mlp_0);1020 struct ggml_tensor * mlp_2 = ggml_mul_mat(ctx0, model.mm_model_mlp_2_w, mlp_0);1021 mlp_2 = ggml_add(ctx0, mlp_2, model.mm_model_mlp_2_b);1022 // mlp_2 ne = [2048, 576, 1, 1]1023 // // AVG Pool Layer 2*2, strides = 21024 mlp_2 = ggml_cont(ctx0, ggml_permute(ctx0, mlp_2, 1, 0, 2, 3));1025 // mlp_2 ne = [576, 2048, 1, 1]1026 mlp_2 = ggml_reshape_4d(ctx0, mlp_2, n_patch, n_patch, mlp_2->ne[1], mlp_2->ne[2]);1027 // mlp_2 ne [24, 24, 2048, 1]1028 mlp_2 = ggml_pool_2d(ctx0, mlp_2, GGML_OP_POOL_AVG, 2, 2, 2, 2, 0, 0);1029 // weight ne = [3, 3, 2048, 1]1030 struct ggml_tensor * peg_0 = ggml_conv_2d_dw(ctx0, model.mm_model_peg_0_w, mlp_2, 1, 1, 1, 1, 1, 1);1031 peg_0 = ggml_cont(ctx0, ggml_permute(ctx0, peg_0, 1, 2, 0, 3));1032 peg_0 = ggml_add(ctx0, peg_0, model.mm_model_peg_0_b);1033 mlp_2 = ggml_cont(ctx0, ggml_permute(ctx0, mlp_2, 1, 2, 0, 3));1034 peg_0 = ggml_add(ctx0, peg_0, mlp_2);1035 peg_0 = ggml_reshape_3d(ctx0, peg_0, peg_0->ne[0], peg_0->ne[1] * peg_0->ne[2], peg_0->ne[3]);1036 embeddings = peg_0;1037 }1038 else {1039 GGML_ABORT("fatal error");1040 }1041 }1042 // minicpmv projector1043 else if (ctx->has_minicpmv_projector)1044 {1045 if (ctx->proj_type == PROJECTOR_TYPE_RESAMPLER) {1046 struct ggml_tensor * q = model.mm_model_query;1047 { // layernorm1048 q = ggml_norm(ctx0, q, eps);1049 q = ggml_add(ctx0, ggml_mul(ctx0, q, model.mm_model_ln_q_w), model.mm_model_ln_q_b);1050 }1051 struct ggml_tensor * v = ggml_mul_mat(ctx0, model.mm_model_kv_proj, embeddings);1052 { // layernorm1053 v = ggml_norm(ctx0, v, eps);1054 v = ggml_add(ctx0, ggml_mul(ctx0, v, model.mm_model_ln_kv_w), model.mm_model_ln_kv_b);1055 }1056 struct ggml_tensor * k;1057 { // position1058 // q = ggml_add(ctx0, q, model.mm_model_pos_embed);1059 k = ggml_add(ctx0, v, pos_embed);1060 }1061 1062 { // attention1063 int hidden_size = 4096;1064 const int d_head = 128;1065 int n_head = hidden_size/d_head;1066 int num_query = 96;1067 if (ctx->minicpmv_version == 2) {1068 hidden_size = 4096;1069 n_head = hidden_size/d_head;1070 num_query = 96;1071 }1072 else if (ctx->minicpmv_version == 3) {1073 hidden_size = 3584;1074 n_head = hidden_size/d_head;1075 num_query = 64;1076 }1077 else if (ctx->minicpmv_version == 4) {1078 hidden_size = 3584;1079 n_head = hidden_size/d_head;1080 num_query = 64;1081 }1082 1083 struct ggml_tensor * Q = ggml_add(ctx0, ggml_mul_mat(ctx0, model.mm_model_attn_q_w, q), model.mm_model_attn_q_b);1084 Q = ggml_scale_inplace(ctx0, Q, 1.0f / sqrt((float)d_head));1085 struct ggml_tensor * K = ggml_add(ctx0, ggml_mul_mat(ctx0, model.mm_model_attn_k_w, k), model.mm_model_attn_k_b);1086 struct ggml_tensor * V = ggml_add(ctx0, ggml_mul_mat(ctx0, model.mm_model_attn_v_w, v), model.mm_model_attn_v_b);1087 // permute1088 Q = ggml_reshape_4d(ctx0, Q, d_head, n_head, num_query, batch_size);1089 Q = ggml_cont(ctx0, ggml_permute(ctx0, Q, 0, 2, 1, 3));1090 Q = ggml_reshape_3d(ctx0, Q, d_head, num_query, n_head * batch_size);1091 K = ggml_reshape_4d(ctx0, K, d_head, n_head, num_positions, batch_size);1092 K = ggml_cont(ctx0, ggml_permute(ctx0, K, 0, 2, 1, 3));1093 K = ggml_reshape_3d(ctx0, K, d_head, num_positions, n_head * batch_size);1094 V = ggml_reshape_4d(ctx0, V, d_head, n_head, num_positions, batch_size);1095 V = ggml_cont(ctx0, ggml_permute(ctx0, V, 1, 2, 0, 3));1096 V = ggml_reshape_3d(ctx0, V, num_positions, d_head, n_head * batch_size);1097 struct ggml_tensor * KQ = ggml_mul_mat(ctx0, K, Q);1098 KQ = ggml_soft_max_inplace(ctx0, KQ);1099 struct ggml_tensor * KQV = ggml_mul_mat(ctx0, V, KQ);1100 KQV = ggml_reshape_4d(ctx0, KQV, d_head, num_query, n_head, batch_size);1101 KQV = ggml_permute(ctx0, KQV, 0, 2, 1, 3);1102 KQV = ggml_cont_3d(ctx0, KQV, hidden_size, num_query, batch_size);1103 1104 embeddings = ggml_add(ctx0, ggml_mul_mat(ctx0, model.mm_model_attn_o_w, KQV), model.mm_model_attn_o_b);1105 }1106 { // layernorm1107 embeddings = ggml_norm(ctx0, embeddings, eps);1108 embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_model_ln_post_w), model.mm_model_ln_post_b);1109 }1110 embeddings = ggml_mul_mat(ctx0, model.mm_model_proj, embeddings);1111 }1112 else {1113 GGML_ASSERT(false);1114 }1115 }1116 // glm projector1117 else if (ctx->has_glm_projector) {1118 if (ctx->proj_type == PROJECTOR_TYPE_GLM_EDGE) {1119 size_t gridsz = (size_t)sqrt(embeddings->ne[1]);1120 embeddings = ggml_cont(ctx0, ggml_permute(ctx0,embeddings,1,0,2,3));1121 embeddings = ggml_reshape_3d(ctx0, embeddings, gridsz, gridsz, embeddings->ne[1]);1122 embeddings = ggml_conv_2d(ctx0, model.mm_model_adapter_conv_w, embeddings, 2, 2, 0, 0, 1, 1);1123 embeddings = ggml_reshape_3d(ctx0, embeddings,embeddings->ne[0]*embeddings->ne[1] , embeddings->ne[2], batch_size);1124 embeddings = ggml_cont(ctx0, ggml_permute(ctx0,embeddings, 1, 0, 2, 3));1125 embeddings = ggml_add(ctx0, embeddings, model.mm_model_adapter_conv_b);1126 //GLU1127 {1128 embeddings = ggml_mul_mat(ctx0, model.mm_model_mlp_0_w, embeddings);1129 embeddings = ggml_norm(ctx0, embeddings, eps);1130 embeddings = ggml_add(ctx0, ggml_mul(ctx0, embeddings, model.mm_model_ln_q_w), model.mm_model_ln_q_b);1131 embeddings = ggml_gelu_inplace(ctx0, embeddings);1132 struct ggml_tensor * x = embeddings;1133 embeddings = ggml_mul_mat(ctx0, model.mm_model_mlp_2_w, embeddings);1134 x = ggml_mul_mat(ctx0, model.mm_model_mlp_1_w,x);1135 embeddings = ggml_silu_inplace(ctx0, embeddings);1136 embeddings = ggml_mul(ctx0, embeddings,x);1137 embeddings = ggml_mul_mat(ctx0, model.mm_model_mlp_3_w, embeddings);1138 }1139 } else {1140 GGML_ABORT("fatel error");1141 }1142 } else if (ctx->proj_type == PROJECTOR_TYPE_MERGER) {1143 embeddings = ggml_reshape_3d(ctx0, embeddings, hidden_size * 4, num_positions / 4, batch_size);1144 1145 embeddings = ggml_mul_mat(ctx0, model.mm_0_w, embeddings);1146 embeddings = ggml_add(ctx0, embeddings, model.mm_0_b);1147 1148 // GELU activation1149 embeddings = ggml_gelu(ctx0, embeddings);1150 1151 // Second linear layer1152 embeddings = ggml_mul_mat(ctx0, model.mm_1_w, embeddings);1153 embeddings = ggml_add(ctx0, embeddings, model.mm_1_b);1154 }1155 1156 // build the graph1157 ggml_build_forward_expand(gf, embeddings);1158 1159 ggml_free(ctx0);1160 1161 return gf;1162}1163 1164// read and create ggml_context containing the tensors and their data1165struct clip_ctx * clip_model_load(const char * fname, const int verbosity = 1) {1166 struct ggml_context * meta = NULL;1167 1168 struct gguf_init_params params = {1169 /*.no_alloc = */ true,1170 /*.ctx = */ &meta,1171 };1172 1173 struct gguf_context * ctx = gguf_init_from_file(fname, params);1174 if (!ctx) {1175 throw std::runtime_error(format("%s: failed to load CLIP model from %s. Does this file exist?\n", __func__, fname));1176 }1177 1178 if (verbosity >= 1) {1179 const int n_tensors = gguf_get_n_tensors(ctx);1180 const int n_kv = gguf_get_n_kv(ctx);1181 const int ftype = get_u32(ctx, KEY_FTYPE);1182 const std::string ftype_str = get_ftype(ftype);1183 const int idx_desc = get_key_idx(ctx, KEY_DESCRIPTION);1184 const std::string description = gguf_get_val_str(ctx, idx_desc);1185 const int idx_name = gguf_find_key(ctx, KEY_NAME);1186 if (idx_name != -1) { // make name optional temporarily as some of the uploaded models missing it due to a bug1187 const std::string name = gguf_get_val_str(ctx, idx_name);1188 LOG_INF("%s: model name: %s\n", __func__, name.c_str());1189 }1190 LOG_INF("%s: description: %s\n", __func__, description.c_str());1191 LOG_INF("%s: GGUF version: %d\n", __func__, gguf_get_version(ctx));1192 LOG_INF("%s: alignment: %zu\n", __func__, gguf_get_alignment(ctx));1193 LOG_INF("%s: n_tensors: %d\n", __func__, n_tensors);1194 LOG_INF("%s: n_kv: %d\n", __func__, n_kv);1195 LOG_INF("%s: ftype: %s\n", __func__, ftype_str.c_str());1196 LOG_INF("\n");1197 }1198 const int n_tensors = gguf_get_n_tensors(ctx);1199 1200 // kv