Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
clip.h134 linesDownload Raw Back to mtmd
1#pragma once2 3#include "ggml.h"4#include "mtmd.h"5 6#include <stddef.h>7#include <stdint.h>8 9#include <map>10 11// !!! Internal header, to be used by mtmd only !!!12 13#define MTMD_INTERNAL_HEADER14 15struct clip_ctx;16 17struct clip_image_size {18    int width;19    int height;20    bool operator==(const clip_image_size & other) const {21        return width == other.width && height == other.height;22    }23    bool operator!=(const clip_image_size & other) const {24        return !(*this == other);25    }26    int area() const {27        // avoid overflow when computing area28        GGML_ASSERT(width  >= 0 && width  <= 46000);29        GGML_ASSERT(height >= 0 && height <= 46000);30        return width * height;31    }32};33 34struct clip_image_f32;35struct clip_image_f32_batch;36 37enum clip_modality {38    CLIP_MODALITY_VISION,39    CLIP_MODALITY_AUDIO,40    CLIP_MODALITY_GEN_AUDIO,41};42 43enum clip_flash_attn_type {44    CLIP_FLASH_ATTN_TYPE_AUTO     = -1,45    CLIP_FLASH_ATTN_TYPE_DISABLED = 0,46    CLIP_FLASH_ATTN_TYPE_ENABLED  = 1,47};48 49struct clip_context_params {50    bool use_gpu;51    enum clip_flash_attn_type flash_attn_type;52    int image_min_tokens;53    int image_max_tokens;54    bool warmup;55    ggml_backend_sched_eval_callback cb_eval;56    void * cb_eval_user_data;57    bool no_alloc;58    mtmd_progress_callback progress_callback;59    void * progress_callback_user_data;60};61 62struct clip_init_result {63    struct clip_ctx * ctx_v; // vision context64    struct clip_ctx * ctx_a; // audio context65    struct clip_ctx * ctx_gen_a; // audio generation context66};67 68struct clip_init_result clip_init(const char * fname, struct clip_context_params ctx_params);69 70void clip_free(struct clip_ctx * ctx);71 72// TODO: should be enum, not string73const char * clip_patch_merge_type(const struct clip_ctx * ctx);74 75int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img);76 77// for M-RoPE, this will be the number of token positions in X and Y directions78// for other models, X will be the total number of tokens and Y will be 179int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img);80int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img);81 82// this should be equal to the embedding dimension of the text model83int clip_n_mmproj_embd(const struct clip_ctx * ctx);84 85// TODO: remove clip_image_encode() and always use batched version86bool clip_image_encode      (struct clip_ctx * ctx, int n_threads, const clip_image_f32 * img, std::vector<float> & out_vec);87bool clip_image_batch_encode(struct clip_ctx * ctx, int n_threads, const struct clip_image_f32_batch * imgs, std::vector<float> & out_batch_embd);88 89enum clip_gen_process_type {90    CLIP_GEN_PROCESS_GEN_UNKNOWN,91    CLIP_GEN_PROCESS_GEN_CODE, // h_state to codes92    CLIP_GEN_PROCESS_GEN_WAV,  // codes to raw PCM audio93};94struct clip_encode_params {95    int n_threads = 1;96    const clip_image_f32_batch * imgs = nullptr;97    std::vector<float> * out_embd = nullptr;98 99    // for audio gen, imgs has exactly one entry: hidden state from backbone (GEN_CODE) or unused (GEN_WAV)100    clip_gen_process_type gen_process = CLIP_GEN_PROCESS_GEN_UNKNOWN;101 102    // GEN_CODE: out_embd receives the embd to feed back to the backbone103    int32_t code0 = 0; // semantic code sampled by the backbone104    int32_t top_k = 50;105    float   top_p = 1.0f;106    std::vector<int32_t> * out_codes = nullptr; // this frame's 16 sampled codes107 108    // GEN_WAV109    const std::vector<int32_t> * codes = nullptr;     // this frame's 16 RVQ codes110    std::vector<float> * out_audio = nullptr;         // decoded PCM samples, F32111    const std::vector<uint8_t> * state_in  = nullptr; // state from previous call, null or wrong size means cold start112    std::vector<uint8_t> *       state_out = nullptr; // state for the next call113};114bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params);115 116bool clip_is_llava(const struct clip_ctx * ctx);117// note for contributor: this clip_is_(model) pattern is deprecated118//                       do NOT add new functions like this119 120bool clip_has_vision_encoder(const struct clip_ctx * ctx);121bool clip_has_audio_encoder(const struct clip_ctx * ctx);122 123bool clip_support_batch(const struct clip_ctx * ctx);124 125int clip_model_n_temporal_merge(const struct clip_ctx * ctx); // TODO @ngxson : remove, refactor this126 127std::map<ggml_backend_dev_t, size_t> clip_get_mem_usage(const struct clip_ctx * ctx);128 129struct clip_cap {130    bool has_vision;131    bool has_audio;132};133struct clip_cap clip_get_cap(const char * fname);134 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai