Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
mtmd.h493 linesDownload Raw Back to mtmd
1#ifndef MTMD_H2#define MTMD_H3 4#include "ggml.h"5#include "llama.h"6 7#include <stddef.h>8#include <stdint.h>9#include <stdbool.h>10 11#ifdef __cplusplus12#include <map>13#include <string>14#include <vector>15#include <cinttypes>16#include <memory>17#endif18 19/**20 * libmtmd: A library for multimodal support in llama.cpp.21 *22 * WARNING: This API is experimental and subject to many BREAKING CHANGES.23 *          Issues related to API usage may receive lower priority support.24 *25 * For the usage, see an example in mtmd-cli.cpp26 *27 * For contributors:28 * - Make sure the C API is aligned with the libllama C API (as in llama.h)29 * - Do not include model name (e.g., qwen, gemma) in the API, use generic terms instead30 * - Keep the API minimal, do not expose internal details unless necessary31 *32 * IMPORTANT: The mtmd module does NOT accept pull requests that are fully or predominantly AI-generated.33 * We encourage human contributors to ensure the quality and reliability of the codebase.34 */35 36#ifdef LLAMA_SHARED37#    if defined(_WIN32) && !defined(__MINGW32__)38#        ifdef LLAMA_BUILD39#            define MTMD_API __declspec(dllexport)40#        else41#            define MTMD_API __declspec(dllimport)42#        endif43#    else44#        define MTMD_API __attribute__ ((visibility ("default")))45#    endif46#else47#    define MTMD_API48#endif49 50#ifdef __cplusplus51extern "C" {52#endif53 54enum mtmd_input_chunk_type {55    MTMD_INPUT_CHUNK_TYPE_TEXT,56    MTMD_INPUT_CHUNK_TYPE_IMAGE,57    MTMD_INPUT_CHUNK_TYPE_AUDIO,58    MTMD_INPUT_CHUNK_TYPE_COUNT, // for validation59};60 61// opaque types62struct mtmd_context;63struct mtmd_bitmap;64struct mtmd_image_tokens;65struct mtmd_input_chunk;66struct mtmd_input_chunks;67struct mtmd_batch;68 69struct mtmd_input_text {70    const char * text;71    size_t text_len;72    bool add_special;73    bool parse_special;74};75 76//77// C API78//79 80typedef struct mtmd_context      mtmd_context;81typedef struct mtmd_bitmap       mtmd_bitmap;82typedef struct mtmd_image_tokens mtmd_image_tokens;83typedef struct mtmd_input_chunk  mtmd_input_chunk;84typedef struct mtmd_input_chunks mtmd_input_chunks;85typedef struct mtmd_input_text   mtmd_input_text;86typedef struct mtmd_batch        mtmd_batch;87 88typedef bool (*mtmd_progress_callback)(float progress, void * user_data);89 90struct mtmd_context_params {91    bool use_gpu;92    bool print_timings;93    int n_threads;94    const char * image_marker; // deprecated, use media_marker instead95    const char * media_marker;96    enum llama_flash_attn_type flash_attn_type;97    bool warmup; // whether to run a warmup encode pass after initialization98 99    // limit number of image tokens, only for vision models with dynamic resolution100    int image_min_tokens; // minimum number of tokens for image input (default: read from metadata)101    int image_max_tokens; // maximum number of tokens for image input (default: read from metadata)102 103    // callback function passed over to mtmd proper104    ggml_backend_sched_eval_callback cb_eval;105    void * cb_eval_user_data;106 107    // batching params108    int32_t batch_max_tokens; // maximum number of output tokens in a batch109                              // (note: this is not a hard-limit, the first image will always be added even if it exceeds this limit)110                              // (default: 1024)111 112    // Called with a progress value between 0.0 and 1.0. Pass NULL to disable.113    // If the provided progress_callback returns true, model loading continues.114    // If it returns false, model loading is immediately aborted.115    mtmd_progress_callback progress_callback;116    void * progress_callback_user_data;117};118 119MTMD_API const char * mtmd_default_marker(void);120 121MTMD_API struct mtmd_context_params mtmd_context_params_default(void);122 123// initialize the mtmd context124// return nullptr on failure125MTMD_API mtmd_context * mtmd_init_from_file(const char * mmproj_fname,126                                            const struct llama_model * text_model,127                                            const struct mtmd_context_params ctx_params);128 129MTMD_API void mtmd_free(mtmd_context * ctx);130 131// whether we need to set non-causal mask before llama_decode132// if chunk is nullptr, we assume the default case where chunk is an image chunk133MTMD_API bool mtmd_decode_use_non_causal(const mtmd_context * ctx, const mtmd_input_chunk * chunk);134 135// whether the current model use M-RoPE for llama_decode136MTMD_API bool mtmd_decode_use_mrope(const mtmd_context * ctx);137 138// whether the current model supports vision input139MTMD_API bool mtmd_support_vision(const mtmd_context * ctx);140 141// whether the current model supports audio input142MTMD_API bool mtmd_support_audio(const mtmd_context * ctx);143 144// get audio sample rate in Hz, for example 16000 for Whisper145// return -1 if audio is not supported146MTMD_API int mtmd_get_audio_sample_rate(const mtmd_context * ctx);147 148// get the current marker string149MTMD_API const char * mtmd_get_marker(const mtmd_context * ctx);150 151// mtmd_bitmap152//153// if bitmap is image:154//     length of data must be nx * ny * 3155//     the data is in RGBRGBRGB... format156//     note: some video-capable models (i.e. qwen-vl) can merge consecutive bitmaps157//           into one chunk, mtmd_tokenize() will automatically handle this158// if bitmap is audio:159//     length of data must be n_samples * sizeof(float)160//     the data is in float format (PCM F32)161//162// if data == nullptr:163//     the bitmap is considered "empty", and will be treated as a placeholder for counting tokens164//     you can pass the bitmap via mtmd_tokenize(), then call mtmd_*_get_n_tokens() to count the tokens165//     note: passing a placeholder bitmap to mtmd_encode() will return an error166MTMD_API mtmd_bitmap *         mtmd_bitmap_init           (uint32_t nx, uint32_t ny, const unsigned char * data);167MTMD_API mtmd_bitmap *         mtmd_bitmap_init_from_audio(size_t n_samples,         const float         * data);168MTMD_API uint32_t              mtmd_bitmap_get_nx     (const mtmd_bitmap * bitmap);169MTMD_API uint32_t              mtmd_bitmap_get_ny     (const mtmd_bitmap * bitmap);170MTMD_API const unsigned char * mtmd_bitmap_get_data   (const mtmd_bitmap * bitmap);171MTMD_API size_t                mtmd_bitmap_get_n_bytes(const mtmd_bitmap * bitmap);172MTMD_API bool                  mtmd_bitmap_is_audio   (const mtmd_bitmap * bitmap);173MTMD_API void                  mtmd_bitmap_free       (mtmd_bitmap * bitmap);174// bitmap ID is optional, but useful for KV cache tracking175// these getters/setters are dedicated functions, so you can for example calculate the hash of the image based on mtmd_bitmap_get_data()176MTMD_API const char * mtmd_bitmap_get_id(const mtmd_bitmap * bitmap);177MTMD_API void         mtmd_bitmap_set_id(mtmd_bitmap * bitmap, const char * id);178 179// mtmd_bitmap lazy180//181// this is a special bitmap that:182// - does not hold the actual data183// - can be expanded into one or more chunks (either media to text chunks)184// user must provide a callback to fill in the data when mtmd_tokenize() is called185// this is useful for large video inputs:186// - allow reading video frame by frame, without loading the entire video into memory187// - allow tracking the whole video with a single ID (for example, the file hash)188 189// set (*out_bitmap) to non-nullptr to emit a bitmap chunk; it will be freed automatically190// set (*out_text) to non-nullptr to emit a text chunk; it must be heap-allocated, null-terminated and will be freed automatically191// either out_bitmap or out_text can be set, but not both192// out_bitmap cannot be another lazy bitmap (no nested lazy allowed)193// return value:194//    0 on success195//   -1 on EOF (signal to mtmd_tokenize to move on)196//   -2 on error (signal to mtmd_tokenize to abort)197typedef int(* mtmd_bitmap_lazy_callback)(198    size_t chunk_idx,199    void * user_data,200    mtmd_bitmap ** out_bitmap,201    char ** out_text);202 203MTMD_API mtmd_bitmap * mtmd_bitmap_init_lazy(mtmd_context * ctx,204                                             const char * id, // usually set to file hash205                                             void * user_data,206                                             mtmd_bitmap_lazy_callback callback);207 208// mtmd_input_chunks209//210// this is simply a list of mtmd_input_chunk211// the elements can only be populated via mtmd_tokenize()212MTMD_API mtmd_input_chunks *      mtmd_input_chunks_init(void);213MTMD_API size_t                   mtmd_input_chunks_size(const mtmd_input_chunks * chunks);214MTMD_API const mtmd_input_chunk * mtmd_input_chunks_get (const mtmd_input_chunks * chunks, size_t idx);215MTMD_API void                     mtmd_input_chunks_free(mtmd_input_chunks * chunks);216 217// mtmd_input_chunk218//219// the instance will be constructed via mtmd_tokenize()220// it will be freed along with mtmd_input_chunks221MTMD_API enum mtmd_input_chunk_type mtmd_input_chunk_get_type        (const mtmd_input_chunk * chunk);222MTMD_API const llama_token *        mtmd_input_chunk_get_tokens_text (const mtmd_input_chunk * chunk, size_t * n_tokens_output);223MTMD_API const mtmd_image_tokens *  mtmd_input_chunk_get_tokens_image(const mtmd_input_chunk * chunk);224MTMD_API size_t                     mtmd_input_chunk_get_n_tokens    (const mtmd_input_chunk * chunk);225// returns nullptr for ID on text chunk226MTMD_API const char *               mtmd_input_chunk_get_id          (const mtmd_input_chunk * chunk);227// number of temporal positions (equals to max(t,h,w) for M-RoPE; equals to n_tokens otherwise)228MTMD_API llama_pos                  mtmd_input_chunk_get_n_pos       (const mtmd_input_chunk * chunk);229 230// in case you want to use custom logic to handle the chunk (i.e. KV cache management)231// you can move the chunk ownership to your own code by copying it232// remember to free the chunk when you are done with it233MTMD_API mtmd_input_chunk * mtmd_input_chunk_copy(const mtmd_input_chunk * chunk);234MTMD_API void               mtmd_input_chunk_free(mtmd_input_chunk * chunk);235 236// save/load an input chunk to/from a buffer (useful for KV save/load)237// important: only chunk's metadata will be saved, the actual image/audio data will not be saved238// the loaded chunk will always be a placeholder, cannot be used for mtmd_encode() or mtmd_batch_encode()239// out_buf can be nullptr (to query expected_out_len)240// returns 0 on success, non-zero on failure241MTMD_API int32_t            mtmd_input_chunk_save(const mtmd_input_chunk * chunk, char * out_buf, size_t out_len, size_t * expected_out_len);242// returns nullptr on failure243MTMD_API mtmd_input_chunk * mtmd_input_chunk_load(const char * buf, size_t len);244 245 246// mtmd_image_tokens247//248// the instance will be constructed via mtmd_tokenize()249// it will be freed along with mtmd_input_chunk250MTMD_API size_t       mtmd_image_tokens_get_n_tokens(const mtmd_image_tokens * image_tokens); // TODO: deprecate251MTMD_API const char * mtmd_image_tokens_get_id      (const mtmd_image_tokens * image_tokens); // TODO: deprecate252// number of temporal positions (equals to max(t,h,w) for M-RoPE; equals to n_tokens otherwise)253MTMD_API llama_pos    mtmd_image_tokens_get_n_pos   (const mtmd_image_tokens * image_tokens); // TODO: deprecate254 255DEPRECATED(MTMD_API size_t mtmd_image_tokens_get_nx(const mtmd_image_tokens * image_tokens),256           "use mtmd_image_tokens_get_decoder_pos() instead");257DEPRECATED(MTMD_API size_t mtmd_image_tokens_get_ny(const mtmd_image_tokens * image_tokens),258           "use mtmd_image_tokens_get_decoder_pos() instead");259 260struct mtmd_decoder_pos {261    uint32_t t;262    uint32_t x;263    uint32_t y;264    uint32_t z; // unused for now, reserved for future use265};266// get position for decoder attention, to be used by M-RoPE models267// i is the index of the embedding token, ranging from 0 to mtmd_image_tokens_get_n_tokens() - 1268// pos_0 is the absolute position of the first token269// return relative position (for example, embedding 0 will have position (0, 0, 0); remember to adjust it to the current absolute position)270MTMD_API struct mtmd_decoder_pos mtmd_image_tokens_get_decoder_pos(const mtmd_image_tokens * image_tokens, llama_pos pos_0, size_t i);271 272// tokenize an input text prompt and a list of bitmaps (images/audio)273// the prompt must have the input image marker (default: "<__media__>") in it274// the default marker is defined by mtmd_default_marker()275// the marker will be replaced with the image/audio chunk276// for example:277//   "here is an image: <__media__>\ndescribe it in detail."278//   this will gives 3 chunks:279//   1. "here is an image: <start_of_image>"280//   2. (image/audio tokens)281//   3. "<end_of_image>\ndescribe it in detail."282// number of bitmaps must be equal to the number of markers in the prompt283// this function is thread-safe (shared ctx)284// return values:285//   0 on success286//   1 on number of bitmaps not matching the number of markers287//   2 on image preprocessing error288MTMD_API int32_t mtmd_tokenize(mtmd_context * ctx,289                               mtmd_input_chunks * output,290                               const mtmd_input_text * text,291                               const mtmd_bitmap ** bitmaps,292                               size_t n_bitmaps);293 294DEPRECATED(MTMD_API int32_t mtmd_encode(mtmd_context * ctx, const mtmd_image_tokens * image_tokens),295           "use mtmd_encode_chunk() instead");296 297// text chunk will be ignored silently, only media chunk will be encoded298// returns 0 on success299// returns 1 on generic error300MTMD_API int32_t mtmd_encode_chunk(mtmd_context * ctx,301                                   const mtmd_input_chunk * chunk);302 303// get output embeddings from the last encode pass304// the reading size (in bytes) is equal to:305// llama_model_n_embd_inp(model) * mtmd_input_chunk_get_n_tokens(chunk) * sizeof(float)306MTMD_API float * mtmd_get_output_embd(mtmd_context * ctx);307 308 309// batch encoding API310// chunks are not owned by the batch, they will not be freed by mtmd_batch_free()311// batch is valid for a given context, cannot be shared across contexts312MTMD_API mtmd_batch * mtmd_batch_init(mtmd_context * ctx);313MTMD_API void         mtmd_batch_free(mtmd_batch * batch);314 315// only media chunks are allowed, text chunks will be rejected316// returns 0 on success317// returns 1 on generic error318// returns 2 if the batch is too large (chunk won't be added)319// returns 3 if it cannot be batched with the existing chunks in the batch320MTMD_API int32_t mtmd_batch_add_chunk(mtmd_batch * batch, const mtmd_input_chunk * chunk);321 322// returns 0 on success323// returns 1 on generic error324MTMD_API int32_t mtmd_batch_encode(mtmd_batch * batch);325MTMD_API float * mtmd_batch_get_output_embd(mtmd_batch * batch, const mtmd_input_chunk * chunk);326 327 328// Set callback for all future logging events.329// If this is not called, or NULL is supplied, everything is output on stderr.330MTMD_API void mtmd_log_set(ggml_log_callback log_callback, void * user_data);331 332// EXPERIMENTAL API to get mmproj's capabilities without initializing the full context333// This is only intended to be used by llama-server, breaking changes is expected334struct mtmd_caps {335    bool inp_vision;336    bool inp_audio;337};338MTMD_API struct mtmd_caps mtmd_get_cap_from_file(const char * mmproj_fname);339 340/////////////////////////////////////////341// EXPERIMENTAL API for audio generation, subjected to breaking changes342 343// represent the pipeline type344enum mtmd_gen_audio_type {345    MTMD_GEN_AUDIO_TYPE_NONE, // not supported346    MTMD_GEN_AUDIO_TYPE_QWEN3TTS,347};348struct mtmd_gen_audio_info {349    enum mtmd_gen_audio_type type;350    int32_t sample_rate; // in Hz, for example 24000 for qwen3tts351};352MTMD_API struct mtmd_gen_audio_info mtmd_gen_audio_get_info(const mtmd_context * ctx);353 354enum mtmd_gen_process_type {355    MTMD_GEN_PROCESS_TYPE_GEN_CODE, // h_state to semantic (codes, mel-spectrogram, etc.)356    MTMD_GEN_PROCESS_TYPE_GEN_WAV,  // convert semantic to PCM audio357                                    // for qwen3tts, this is code2wav358};359struct mtmd_gen_inp {360    enum mtmd_gen_process_type type;361 362    // for MTMD_GEN_PROCESS_TYPE_GEN_CODE363    int32_t code0;  // the sampled codebook 0 entry from backbone364    float * embd;   // the hidden state from backbone, must have n_text_embd elements365    int32_t top_k;366    float   top_p;367 368    // for MTMD_GEN_PROCESS_TYPE_GEN_WAV369    int32_t * codes;370    size_t    n_codes;371    const char * state_data;372    size_t       state_size;373};374struct mtmd_gen_out {375    // note: output memory is allocated by the context, valid until next process() call376 377    // for MTMD_GEN_PROCESS_TYPE_GEN_CODE378    const int32_t * codes;379    size_t n_codes;380    const float * embd; // the generated hidden state, to be fed back to backbone381                        // it must have n_text_embd elements382 383    // for MTMD_GEN_PROCESS_TYPE_GEN_WAV384    const float * audio;385    size_t        n_samples;386    const char * state_data;387    size_t       state_size;388};389// note: this API is stateless, caller must handle state management and audio frame accumulation390MTMD_API int32_t mtmd_gen_audio_process(mtmd_context * ctx,391                                const struct mtmd_gen_inp * inp,392                                struct mtmd_gen_out * out);393 394/////////////////////////////////////////395 396// test function, to be used in test-mtmd-c-api.c397MTMD_API mtmd_input_chunks * mtmd_test_create_input_chunks(void);398 399#ifdef __cplusplus400} // extern "C"401#endif402 403// Get memory usage of the current model in bytes, per backend device404// Note: this is an unstable API, used internally by fit_params; it WILL be removed or changed without deprecation405#ifdef __cplusplus406MTMD_API std::map<ggml_backend_dev_t, size_t> mtmd_get_memory_usage(407    const char * mmproj_fname,408    struct mtmd_context_params ctx_params);409#endif410 411//412// C++ wrappers413//414 415#ifdef __cplusplus416 417namespace mtmd {418 419struct mtmd_context_deleter {420    void operator()(mtmd_context * val) { mtmd_free(val); }421};422using context_ptr = std::unique_ptr<mtmd_context, mtmd_context_deleter>;423 424struct mtmd_bitmap_deleter {425    void operator()(mtmd_bitmap * val) { mtmd_bitmap_free(val); }426};427using bitmap_ptr = std::unique_ptr<mtmd_bitmap, mtmd_bitmap_deleter>;428 429struct mtmd_input_chunks_deleter {430    void operator()(mtmd_input_chunks * val) { mtmd_input_chunks_free(val); }431};432using input_chunks_ptr = std::unique_ptr<mtmd_input_chunks, mtmd_input_chunks_deleter>;433 434struct mtmd_input_chunk_deleter {435    void operator()(mtmd_input_chunk * val) { mtmd_input_chunk_free(val); }436};437using input_chunk_ptr = std::unique_ptr<mtmd_input_chunk, mtmd_input_chunk_deleter>;438 439struct mtmd_batch_deleter {440    void operator()(mtmd_batch * val) { mtmd_batch_free(val); }441};442using batch_ptr = std::unique_ptr<mtmd_batch, mtmd_batch_deleter>;443 444struct bitmap {445    bitmap_ptr ptr;446    bitmap() : ptr(nullptr) {}447    bitmap(mtmd_bitmap * bitmap) : ptr(bitmap) {}448    bitmap(bitmap && other) noexcept : ptr(std::move(other.ptr)) {}449    bitmap(uint32_t nx, uint32_t ny, const unsigned char * data) {450        ptr.reset(mtmd_bitmap_init(nx, ny, data));451    }452    ~bitmap() = default;453    uint32_t nx() const { return mtmd_bitmap_get_nx(ptr.get()); }454    uint32_t ny() const { return mtmd_bitmap_get_ny(ptr.get()); }455    const unsigned char * data() const { return mtmd_bitmap_get_data(ptr.get()); }456    size_t n_bytes() const { return mtmd_bitmap_get_n_bytes(ptr.get()); }457    std::string id() const { return mtmd_bitmap_get_id(ptr.get()); }458    void set_id(const char * id) const { mtmd_bitmap_set_id(ptr.get(), id); }459};460 461struct bitmaps {462    std::vector<bitmap> entries;463    ~bitmaps() = default;464    // return list of pointers to mtmd_bitmap465    // example:466    //   auto bitmaps_c_ptr = bitmaps.c_ptr();467    //   int32_t res = mtmd_tokenize(... bitmaps_c_ptr.data(), bitmaps_c_ptr.size());468    std::vector<const mtmd_bitmap *> c_ptr() {469        std::vector<const mtmd_bitmap *> res(entries.size());470        for (size_t i = 0; i < entries.size(); i++) {471            res[i] = entries[i].ptr.get();472        }473        return res;474    }475};476 477struct input_chunks {478    input_chunks_ptr ptr;479    input_chunks() = default;480    input_chunks(mtmd_input_chunks * chunks) : ptr(chunks) {}481    ~input_chunks() = default;482    size_t size() const { return mtmd_input_chunks_size(ptr.get()); }483    const mtmd_input_chunk * operator[](size_t idx) const {484        return mtmd_input_chunks_get(ptr.get(), idx);485    }486};487 488} // namespace mtmd489 490#endif491 492#endif493