Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#pragma once2 3#include "ggml.h"4#include "mtmd.h"5 6#include <stddef.h>7#include <stdint.h>8 9#include <map>10 11// !!! Internal header, to be used by mtmd only !!!12 13#define MTMD_INTERNAL_HEADER14 15struct clip_ctx;16 17struct clip_image_size {18 int width;19 int height;20 bool operator==(const clip_image_size & other) const {21 return width == other.width && height == other.height;22 }23 bool operator!=(const clip_image_size & other) const {24 return !(*this == other);25 }26 int area() const {27 // avoid overflow when computing area28 GGML_ASSERT(width >= 0 && width <= 46000);29 GGML_ASSERT(height >= 0 && height <= 46000);30 return width * height;31 }32};33 34struct clip_image_f32;35struct clip_image_f32_batch;36 37enum clip_modality {38 CLIP_MODALITY_VISION,39 CLIP_MODALITY_AUDIO,40 CLIP_MODALITY_GEN_AUDIO,41};42 43enum clip_flash_attn_type {44 CLIP_FLASH_ATTN_TYPE_AUTO = -1,45 CLIP_FLASH_ATTN_TYPE_DISABLED = 0,46 CLIP_FLASH_ATTN_TYPE_ENABLED = 1,47};48 49struct clip_context_params {50 bool use_gpu;51 enum clip_flash_attn_type flash_attn_type;52 int image_min_tokens;53 int image_max_tokens;54 bool warmup;55 ggml_backend_sched_eval_callback cb_eval;56 void * cb_eval_user_data;57 bool no_alloc;58 mtmd_progress_callback progress_callback;59 void * progress_callback_user_data;60};61 62struct clip_init_result {63 struct clip_ctx * ctx_v; // vision context64 struct clip_ctx * ctx_a; // audio context65 struct clip_ctx * ctx_gen_a; // audio generation context66};67 68struct clip_init_result clip_init(const char * fname, struct clip_context_params ctx_params);69 70void clip_free(struct clip_ctx * ctx);71 72// TODO: should be enum, not string73const char * clip_patch_merge_type(const struct clip_ctx * ctx);74 75int clip_n_output_tokens(const clip_ctx * ctx, const clip_image_f32 * img);76 77// for M-RoPE, this will be the number of token positions in X and Y directions78// for other models, X will be the total number of tokens and Y will be 179int clip_n_output_tokens_x(const clip_ctx * ctx, const clip_image_f32 * img);80int clip_n_output_tokens_y(const clip_ctx * ctx, const clip_image_f32 * img);81 82// this should be equal to the embedding dimension of the text model83int clip_n_mmproj_embd(const struct clip_ctx * ctx);84 85// TODO: remove clip_image_encode() and always use batched version86bool clip_image_encode (struct clip_ctx * ctx, int n_threads, const clip_image_f32 * img, std::vector<float> & out_vec);87bool clip_image_batch_encode(struct clip_ctx * ctx, int n_threads, const struct clip_image_f32_batch * imgs, std::vector<float> & out_batch_embd);88 89enum clip_gen_process_type {90 CLIP_GEN_PROCESS_GEN_UNKNOWN,91 CLIP_GEN_PROCESS_GEN_CODE, // h_state to codes92 CLIP_GEN_PROCESS_GEN_WAV, // codes to raw PCM audio93};94struct clip_encode_params {95 int n_threads = 1;96 const clip_image_f32_batch * imgs = nullptr;97 std::vector<float> * out_embd = nullptr;98 99 // for audio gen, imgs has exactly one entry: hidden state from backbone (GEN_CODE) or unused (GEN_WAV)100 clip_gen_process_type gen_process = CLIP_GEN_PROCESS_GEN_UNKNOWN;101 102 // GEN_CODE: out_embd receives the embd to feed back to the backbone103 int32_t code0 = 0; // semantic code sampled by the backbone104 int32_t top_k = 50;105 float top_p = 1.0f;106 std::vector<int32_t> * out_codes = nullptr; // this frame's 16 sampled codes107 108 // GEN_WAV109 const std::vector<int32_t> * codes = nullptr; // this frame's 16 RVQ codes110 std::vector<float> * out_audio = nullptr; // decoded PCM samples, F32111 const std::vector<uint8_t> * state_in = nullptr; // state from previous call, null or wrong size means cold start112 std::vector<uint8_t> * state_out = nullptr; // state for the next call113};114bool clip_encode(struct clip_ctx * ctx, struct clip_encode_params * params);115 116bool clip_is_llava(const struct clip_ctx * ctx);117// note for contributor: this clip_is_(model) pattern is deprecated118// do NOT add new functions like this119 120bool clip_has_vision_encoder(const struct clip_ctx * ctx);121bool clip_has_audio_encoder(const struct clip_ctx * ctx);122 123bool clip_support_batch(const struct clip_ctx * ctx);124 125int clip_model_n_temporal_merge(const struct clip_ctx * ctx); // TODO @ngxson : remove, refactor this126 127std::map<ggml_backend_dev_t, size_t> clip_get_mem_usage(const struct clip_ctx * ctx);128 129struct clip_cap {130 bool has_vision;131 bool has_audio;132};133struct clip_cap clip_get_cap(const char * fname);134 