Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#pragma once2 3#include "ggml.h"4#include "clip-model.h"5 6#include <vector>7#include <string>8 9#define MTMD_INTERNAL_HEADER10 11struct mtmd_image_preproc_out {12 std::vector<clip_image_f32> entries;13 // grid size is required for llava-uhd style models14 15 clip_image_f32 overview; // overview image (downscaled image)16 int grid_x = 0;17 int grid_y = 0;18 19 void append(const clip_hparams & hparams, const clip_image_u8 & img, bool normalized = true);20 void append(const clip_hparams & hparams, const std::vector<clip_image_u8> & imgs, bool normalized = true);21 void append(const clip_hparams & hparams, clip_image_f32 & img, bool normalized = true);22 23 void append_overview(const clip_hparams & hparams, const clip_image_u8 & img, bool normalized = true);24 bool has_overview() const {25 return overview.nx() > 0 || overview.ny() > 0;26 }27};28 29// base class, models must inherit from this class30struct mtmd_image_preprocessor {31 const clip_hparams & hparams;32 33 mtmd_image_preprocessor(const clip_ctx * ctx): hparams(*clip_get_hparams(ctx)) {}34 35 virtual ~mtmd_image_preprocessor() = default;36 virtual mtmd_image_preproc_out preprocess(const clip_image_u8 & img) = 0;37};38 39/**40 * implementation of LLaVA-UHD:41 * - https://arxiv.org/pdf/2403.1170342 * - https://github.com/thunlp/LLaVA-UHD43 * - https://github.com/thunlp/LLaVA-UHD/blob/302301bc2175f7e717fb8548516188e89f649753/llava_uhd/train/llava-uhd/slice_logic.py#L11844 *45 * overview:46 * - an image always have a single overview (downscaled image)47 * - an image can have 0 or multiple slices, depending on the image size48 * - each slice can then be considered as a separate image49 *50 * note: the term "slice" and "tile" are used interchangeably51 *52 * for example:53 *54 * [overview] --> [slice 1] --> [slice 2]55 * | |56 * +--> [slice 3] --> [slice 4]57 *58 * NOTE: for the ordering of overview, set "ov_img_first" on the mtmd_context59 */60struct mtmd_image_preprocessor_llava_uhd : mtmd_image_preprocessor {61 mtmd_image_preprocessor_llava_uhd(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}62 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;63 64 struct slice_coordinates {65 int x;66 int y;67 clip_image_size size;68 };69 70 struct slice_instructions {71 clip_image_size overview_size; // size of downscaled image72 clip_image_size refined_size; // size of image right before slicing (must be multiple of slice size)73 clip_image_size grid_size; // grid_size.width * grid_size.height = number of slices74 std::vector<slice_coordinates> slices;75 };76 77 virtual slice_instructions get_slice_instructions(const clip_image_size & original_size);78 79 struct slice_output {80 clip_image_u8 overview;81 std::vector<clip_image_u8> slices;82 };83 slice_output slice_image(const clip_image_u8 & img, const slice_instructions & inst);84 85protected:86 clip_image_size get_best_resize(const clip_image_size & original_size, int scale_resolution, int patch_size, bool allow_upscale = false);87 88private:89 clip_image_size resize_maintain_aspect_ratio(const clip_image_size & orig, const clip_image_size & target_max);90 91 /**92 * Selects the best resolution from a list of possible resolutions based on the original size.93 *94 * For example, when given a list of resolutions:95 * - 100x10096 * - 200x10097 * - 100x20098 * - 200x20099 *100 * And an input image of size 111x200, then 100x200 is the best fit (least wasted resolution).101 *102 * @param original_size The original size of the image103 * @param possible_resolutions A list of possible resolutions104 * @return The best fit resolution105 */106 clip_image_size select_best_resolution(const clip_image_size & original_size, const std::vector<clip_image_size> & possible_resolutions);107 int ensure_divide(int length, int patch_size);108 clip_image_size get_refine_size(const clip_image_size & original_size, const clip_image_size & grid, int scale_resolution, int patch_size, bool allow_upscale = false);109 clip_image_size get_best_grid(const int max_slice_nums, const int multiple, const float log_ratio);110};111 112// downscale or upscale the input image to fixed size113struct mtmd_image_preprocessor_fixed_size : mtmd_image_preprocessor {114 mtmd_image_preprocessor_fixed_size(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}115 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;116};117 118// resize image to multiple of patch_size*n_merge, while preserving aspect ratio119// if image_resize_pad is true, the resized image will be padded, otherwise it will be either stretched or center-cropped depending on image_resize_pad120// this is used by models with native support for dynamic image size, for example: Qwen-VL, Pixtral, Kimi-VL, etc121struct mtmd_image_preprocessor_dyn_size : mtmd_image_preprocessor {122 mtmd_image_preprocessor_dyn_size(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}123 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;124};125 126// similar to mtmd_image_preprocessor_dyn_size, but resize the image to have longest edge equal to hparams.image_longest_edge, while preserving aspect ratio127struct mtmd_image_preprocessor_longest_edge : mtmd_image_preprocessor {128 mtmd_image_preprocessor_longest_edge(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}129 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;130};131 132// custom llava-uhd slicing logic for MiniCPM-V133struct mtmd_image_preprocessor_minicpmv : mtmd_image_preprocessor_llava_uhd {134 using mtmd_image_preprocessor_llava_uhd::mtmd_image_preprocessor_llava_uhd;135 slice_instructions get_slice_instructions(const clip_image_size & original_size) override;136};137 138// custom llava-uhd slicing logic for LFM2139// ref: https://github.com/huggingface/transformers/blob/v5.1.0/src/transformers/models/lfm2_vl/image_processing_lfm2_vl_fast.py140struct mtmd_image_preprocessor_lfm2 : mtmd_image_preprocessor_llava_uhd {141 // ref: https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B/blob/main/processor_config.json142 static constexpr int min_tiles = 2;143 static constexpr int max_tiles = 10;144 static constexpr float max_pixels_tolerance = 2.0f;145 static constexpr int tile_size = 512;146 147 using mtmd_image_preprocessor_llava_uhd::mtmd_image_preprocessor_llava_uhd;148 slice_instructions get_slice_instructions(const clip_image_size & original_size) override;149 150private:151 clip_image_size find_closest_aspect_ratio(152 float aspect_ratio,153 const std::vector<clip_image_size> & target_ratios,154 int width, int height);155 std::vector<clip_image_size> get_target_ratios();156 clip_image_size get_grid_layout(int height, int width);157};158 159struct mtmd_image_preprocessor_idefics3 : mtmd_image_preprocessor_llava_uhd {160 mtmd_image_preprocessor_idefics3(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {}161 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;162};163 164struct mtmd_image_preprocessor_internvl : mtmd_image_preprocessor_llava_uhd {165 mtmd_image_preprocessor_internvl(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {}166 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;167};168 169// DeepSeek-OCR (v1/v2) global view + optional local tile grid170struct mtmd_image_preprocessor_deepseekocr : mtmd_image_preprocessor {171 mtmd_image_preprocessor_deepseekocr(const clip_ctx * ctx)172 : mtmd_image_preprocessor(ctx),173 fuse_row(clip_get_projector_type(ctx) == PROJECTOR_TYPE_DEEPSEEKOCR),174 base_size(hparams.image_size),175 tile_size(hparams.preproc_tile_size),176 min_tiles(hparams.preproc_min_tiles),177 max_tiles(hparams.preproc_max_tiles) {}178 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;179 180private:181 bool fuse_row; // v1 fuses a tile-row into one image; v2 keeps tiles separate182 int base_size; // global view183 int tile_size; // each tile184 int min_tiles;185 int max_tiles;186 187 std::vector<clip_image_size> get_target_ratios() const;188 clip_image_size find_closest_aspect_ratio(189 float aspect_ratio,190 const std::vector<clip_image_size> & target_ratios,191 int width, int height) const;192};193 194// custom image preprocessing for Step3VL195// ref: https://huggingface.co/stepfun-ai/Step3-VL-10B/blob/main/processing_step3.py196struct mtmd_image_preprocessor_step3vl : mtmd_image_preprocessor_llava_uhd {197 mtmd_image_preprocessor_step3vl(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {}198 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;199 static slice_instructions build_slice_instructions(const clip_hparams & params, const clip_image_size & prepared_size);200 201private:202 static constexpr int default_image_longest_edge = 3024;203 static constexpr int default_image_crop_size = 504;204 static constexpr float small_aspect_ratio_limit = 1.5f;205 static constexpr float wide_aspect_ratio_limit = 4.0f;206 static constexpr float crop_rounding_threshold = 0.2f;207 208 void img_u8_resize_bilinear_to_f32(209 const clip_image_u8 & src,210 clip_image_f32 & dst,211 int target_width,212 int target_height,213 const float mean[3],214 const float std[3]);215 static int get_image_longest_edge(const clip_hparams & params);216 static int determine_window_size(const clip_hparams & params, int longer, int shorter);217 static int calc_crop_extent(int length, int window_size);218 static std::vector<int> calc_grid(int length, int window_size);219 static clip_image_u8 prepare_image(const clip_image_u8 & img, const clip_hparams & params);220 static clip_image_u8 crop_with_black_padding(const clip_image_u8 & image, int x, int y, int w, int h);221};222 223struct mtmd_image_preprocessor_youtuvl : mtmd_image_preprocessor {224 mtmd_image_preprocessor_youtuvl(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}225 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;226};227 228// similar to llava_uhd, but has add_newline229struct mtmd_image_preprocessor_granite : mtmd_image_preprocessor_llava_uhd {230 mtmd_image_preprocessor_granite(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {}231 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;232};233 234// pick the patch grid closest to the input aspect ratio under the per-image token cap, stretch-resize.235struct mtmd_image_preprocessor_muse_glimmer : mtmd_image_preprocessor {236 mtmd_image_preprocessor_muse_glimmer(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}237 mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;238};239 