Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
mtmd-image.h239 linesDownload Raw Back to mtmd
1#pragma once2 3#include "ggml.h"4#include "clip-model.h"5 6#include <vector>7#include <string>8 9#define MTMD_INTERNAL_HEADER10 11struct mtmd_image_preproc_out {12    std::vector<clip_image_f32> entries;13    // grid size is required for llava-uhd style models14 15    clip_image_f32 overview; // overview image (downscaled image)16    int grid_x = 0;17    int grid_y = 0;18 19    void append(const clip_hparams & hparams, const clip_image_u8 & img, bool normalized = true);20    void append(const clip_hparams & hparams, const std::vector<clip_image_u8> & imgs, bool normalized = true);21    void append(const clip_hparams & hparams, clip_image_f32 & img, bool normalized = true);22 23    void append_overview(const clip_hparams & hparams, const clip_image_u8 & img, bool normalized = true);24    bool has_overview() const {25        return overview.nx() > 0 || overview.ny() > 0;26    }27};28 29// base class, models must inherit from this class30struct mtmd_image_preprocessor {31    const clip_hparams & hparams;32 33    mtmd_image_preprocessor(const clip_ctx * ctx): hparams(*clip_get_hparams(ctx)) {}34 35    virtual ~mtmd_image_preprocessor() = default;36    virtual mtmd_image_preproc_out preprocess(const clip_image_u8 & img) = 0;37};38 39/**40 * implementation of LLaVA-UHD:41 *  - https://arxiv.org/pdf/2403.1170342 *  - https://github.com/thunlp/LLaVA-UHD43 *  - https://github.com/thunlp/LLaVA-UHD/blob/302301bc2175f7e717fb8548516188e89f649753/llava_uhd/train/llava-uhd/slice_logic.py#L11844 *45 * overview:46 *   - an image always have a single overview (downscaled image)47 *   - an image can have 0 or multiple slices, depending on the image size48 *   - each slice can then be considered as a separate image49 *50 * note: the term "slice" and "tile" are used interchangeably51 *52 * for example:53 *54 * [overview] --> [slice 1] --> [slice 2]55 *           |                |56 *           +--> [slice 3] --> [slice 4]57 *58 * NOTE: for the ordering of overview, set "ov_img_first" on the mtmd_context59 */60struct mtmd_image_preprocessor_llava_uhd : mtmd_image_preprocessor {61    mtmd_image_preprocessor_llava_uhd(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}62    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;63 64    struct slice_coordinates {65        int x;66        int y;67        clip_image_size size;68    };69 70    struct slice_instructions {71        clip_image_size overview_size; // size of downscaled image72        clip_image_size refined_size;  // size of image right before slicing (must be multiple of slice size)73        clip_image_size grid_size;     // grid_size.width * grid_size.height = number of slices74        std::vector<slice_coordinates> slices;75    };76 77    virtual slice_instructions get_slice_instructions(const clip_image_size & original_size);78 79    struct slice_output {80        clip_image_u8 overview;81        std::vector<clip_image_u8> slices;82    };83    slice_output slice_image(const clip_image_u8 & img, const slice_instructions & inst);84 85protected:86    clip_image_size get_best_resize(const clip_image_size & original_size, int scale_resolution, int patch_size, bool allow_upscale = false);87 88private:89    clip_image_size resize_maintain_aspect_ratio(const clip_image_size & orig, const clip_image_size & target_max);90 91    /**92     * Selects the best resolution from a list of possible resolutions based on the original size.93     *94     * For example, when given a list of resolutions:95     *  - 100x10096     *  - 200x10097     *  - 100x20098     *  - 200x20099     *100     * And an input image of size 111x200, then 100x200 is the best fit (least wasted resolution).101     *102     * @param original_size The original size of the image103     * @param possible_resolutions A list of possible resolutions104     * @return The best fit resolution105     */106    clip_image_size select_best_resolution(const clip_image_size & original_size, const std::vector<clip_image_size> & possible_resolutions);107    int ensure_divide(int length, int patch_size);108    clip_image_size get_refine_size(const clip_image_size & original_size, const clip_image_size & grid, int scale_resolution, int patch_size, bool allow_upscale = false);109    clip_image_size get_best_grid(const int max_slice_nums, const int multiple, const float log_ratio);110};111 112// downscale or upscale the input image to fixed size113struct mtmd_image_preprocessor_fixed_size : mtmd_image_preprocessor {114    mtmd_image_preprocessor_fixed_size(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}115    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;116};117 118// resize image to multiple of patch_size*n_merge, while preserving aspect ratio119// if image_resize_pad is true, the resized image will be padded, otherwise it will be either stretched or center-cropped depending on image_resize_pad120// this is used by models with native support for dynamic image size, for example: Qwen-VL, Pixtral, Kimi-VL, etc121struct mtmd_image_preprocessor_dyn_size : mtmd_image_preprocessor {122    mtmd_image_preprocessor_dyn_size(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}123    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;124};125 126// similar to mtmd_image_preprocessor_dyn_size, but resize the image to have longest edge equal to hparams.image_longest_edge, while preserving aspect ratio127struct mtmd_image_preprocessor_longest_edge : mtmd_image_preprocessor {128    mtmd_image_preprocessor_longest_edge(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}129    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;130};131 132// custom llava-uhd slicing logic for MiniCPM-V133struct mtmd_image_preprocessor_minicpmv : mtmd_image_preprocessor_llava_uhd {134    using mtmd_image_preprocessor_llava_uhd::mtmd_image_preprocessor_llava_uhd;135    slice_instructions get_slice_instructions(const clip_image_size & original_size) override;136};137 138// custom llava-uhd slicing logic for LFM2139// ref: https://github.com/huggingface/transformers/blob/v5.1.0/src/transformers/models/lfm2_vl/image_processing_lfm2_vl_fast.py140struct mtmd_image_preprocessor_lfm2 : mtmd_image_preprocessor_llava_uhd {141    // ref: https://huggingface.co/LiquidAI/LFM2.5-VL-1.6B/blob/main/processor_config.json142    static constexpr int   min_tiles            = 2;143    static constexpr int   max_tiles            = 10;144    static constexpr float max_pixels_tolerance = 2.0f;145    static constexpr int   tile_size            = 512;146 147    using mtmd_image_preprocessor_llava_uhd::mtmd_image_preprocessor_llava_uhd;148    slice_instructions get_slice_instructions(const clip_image_size & original_size) override;149 150private:151    clip_image_size find_closest_aspect_ratio(152            float aspect_ratio,153            const std::vector<clip_image_size> & target_ratios,154            int width, int height);155    std::vector<clip_image_size> get_target_ratios();156    clip_image_size get_grid_layout(int height, int width);157};158 159struct mtmd_image_preprocessor_idefics3 : mtmd_image_preprocessor_llava_uhd {160    mtmd_image_preprocessor_idefics3(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {}161    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;162};163 164struct mtmd_image_preprocessor_internvl : mtmd_image_preprocessor_llava_uhd {165    mtmd_image_preprocessor_internvl(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {}166    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;167};168 169// DeepSeek-OCR (v1/v2) global view + optional local tile grid170struct mtmd_image_preprocessor_deepseekocr : mtmd_image_preprocessor {171    mtmd_image_preprocessor_deepseekocr(const clip_ctx * ctx)172        : mtmd_image_preprocessor(ctx),173            fuse_row(clip_get_projector_type(ctx) == PROJECTOR_TYPE_DEEPSEEKOCR),174          base_size(hparams.image_size),175          tile_size(hparams.preproc_tile_size),176          min_tiles(hparams.preproc_min_tiles),177          max_tiles(hparams.preproc_max_tiles) {}178    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;179 180private:181    bool fuse_row; // v1 fuses a tile-row into one image; v2 keeps tiles separate182    int base_size; // global view183    int tile_size; // each tile184    int min_tiles;185    int max_tiles;186 187    std::vector<clip_image_size> get_target_ratios() const;188    clip_image_size find_closest_aspect_ratio(189            float aspect_ratio,190            const std::vector<clip_image_size> & target_ratios,191            int width, int height) const;192};193 194// custom image preprocessing for Step3VL195// ref: https://huggingface.co/stepfun-ai/Step3-VL-10B/blob/main/processing_step3.py196struct mtmd_image_preprocessor_step3vl : mtmd_image_preprocessor_llava_uhd {197    mtmd_image_preprocessor_step3vl(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {}198    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;199    static slice_instructions build_slice_instructions(const clip_hparams & params, const clip_image_size & prepared_size);200 201private:202    static constexpr int   default_image_longest_edge = 3024;203    static constexpr int   default_image_crop_size    = 504;204    static constexpr float small_aspect_ratio_limit   = 1.5f;205    static constexpr float wide_aspect_ratio_limit    = 4.0f;206    static constexpr float crop_rounding_threshold    = 0.2f;207 208    void img_u8_resize_bilinear_to_f32(209            const clip_image_u8 & src,210            clip_image_f32 & dst,211            int target_width,212            int target_height,213            const float mean[3],214            const float std[3]);215    static int get_image_longest_edge(const clip_hparams & params);216    static int determine_window_size(const clip_hparams & params, int longer, int shorter);217    static int calc_crop_extent(int length, int window_size);218    static std::vector<int> calc_grid(int length, int window_size);219    static clip_image_u8 prepare_image(const clip_image_u8 & img, const clip_hparams & params);220    static clip_image_u8 crop_with_black_padding(const clip_image_u8 & image, int x, int y, int w, int h);221};222 223struct mtmd_image_preprocessor_youtuvl : mtmd_image_preprocessor {224    mtmd_image_preprocessor_youtuvl(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}225    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;226};227 228// similar to llava_uhd, but has add_newline229struct mtmd_image_preprocessor_granite : mtmd_image_preprocessor_llava_uhd {230    mtmd_image_preprocessor_granite(const clip_ctx * ctx) : mtmd_image_preprocessor_llava_uhd(ctx) {}231    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;232};233 234// pick the patch grid closest to the input aspect ratio under the per-image token cap, stretch-resize.235struct mtmd_image_preprocessor_muse_glimmer : mtmd_image_preprocessor {236    mtmd_image_preprocessor_muse_glimmer(const clip_ctx * ctx) : mtmd_image_preprocessor(ctx) {}237    mtmd_image_preproc_out preprocess(const clip_image_u8 & img) override;238};239 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai