Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1// fix problem with std::min and std::max2#if defined(_WIN32)3#define WIN32_LEAN_AND_MEAN4#ifndef NOMINMAX5# define NOMINMAX6#endif7#include <windows.h>8#endif9 10#include "mtmd.h"11#include "mtmd-helper.h"12#include "mtmd-helper-common.h"13#include "llama.h"14 15#include <algorithm>16#include <cinttypes>17#include <vector>18 19//#define MTMD_AUDIO_DEBUG20 21#define MINIAUDIO_IMPLEMENTATION22#ifndef MTMD_AUDIO_DEBUG23# define MA_NO_ENCODING24#endif25#define MA_NO_DEVICE_IO26#define MA_NO_RESOURCE_MANAGER27#define MA_NO_NODE_GRAPH28#define MA_NO_ENGINE29#define MA_NO_GENERATION30#define MA_API static31#include "miniaudio/miniaudio.h"32 33#define STB_IMAGE_IMPLEMENTATION34#include "stb/stb_image.h"35 36#ifdef MTMD_INTERNAL_HEADER37#error "mtmd-helper is a public library outside of mtmd. it must not include internal headers"38#endif39 40#ifdef MTMD_VIDEO41#include "sheredom/subprocess.h"42#include <thread>43#endif44 45//46// internal logging functions47//48 49void mtmd_helper_log_set(ggml_log_callback log_callback, void * user_data) {50 if (log_callback == nullptr) {51 log_callback = g_logger.default_callback;52 }53 g_logger.log_callback = log_callback;54 g_logger.log_callback_user_data = user_data;55 mtmd_log_set(log_callback, user_data);56}57 58//59// helper functions60//61 62size_t mtmd_helper_get_n_tokens(const mtmd_input_chunks * chunks) {63 size_t n_tokens = 0;64 for (size_t i = 0; i < mtmd_input_chunks_size(chunks); i++) {65 auto chunk = mtmd_input_chunks_get(chunks, i);66 n_tokens += mtmd_input_chunk_get_n_tokens(chunk);67 }68 return n_tokens;69}70 71llama_pos mtmd_helper_get_n_pos(const mtmd_input_chunks * chunks) {72 llama_pos n_pos = 0;73 for (size_t i = 0; i < mtmd_input_chunks_size(chunks); i++) {74 auto chunk = mtmd_input_chunks_get(chunks, i);75 n_pos += mtmd_input_chunk_get_n_pos(chunk);76 }77 return n_pos;78}79 80void mtmd_helper_image_get_decoder_pos(const mtmd_image_tokens * chunks, llama_pos pos_0, mtmd_decoder_pos * out_pos) {81 size_t n_tokens = mtmd_image_tokens_get_n_tokens(chunks);82 for (size_t i = 0; i < n_tokens; i++) {83 out_pos[i] = mtmd_image_tokens_get_decoder_pos(chunks, pos_0, i);84 }85}86 87// Helper class to set non-causal attention via RAII88class scope_non_causal {89public:90 scope_non_causal(llama_context * context, bool enabled) : context_(context), enabled_(enabled) {91 if (enabled_) {92 // TODO @ngxson : need to make sure only one image is processed at a time, and n_ubatch must be enough to hold the image93 llama_set_causal_attn(context_, false);94 }95 }96 ~scope_non_causal() {97 if (enabled_) {98 llama_set_causal_attn(context_, true);99 }100 }101 102 scope_non_causal(const scope_non_causal &) = delete;103 scope_non_causal & operator=(const scope_non_causal &) = delete;104 105private:106 llama_context * context_;107 bool enabled_;108};109 110// Helper function for decoding an image whose embeddings have already been calculated111int32_t mtmd_helper_decode_image_chunk(112 mtmd_context * ctx,113 struct llama_context * lctx,114 const mtmd_input_chunk * chunk,115 float * encoded_embd,116 llama_pos n_past,117 llama_seq_id seq_id,118 int32_t n_batch,119 llama_pos * new_n_past,120 mtmd_helper_post_decode_callback callback,121 void * user_data) {122 GGML_ASSERT(n_batch > 0);123 auto chunk_type = mtmd_input_chunk_get_type(chunk);124 const char * name = chunk_type == MTMD_INPUT_CHUNK_TYPE_IMAGE ? "image" : "audio";125 if (chunk_type == MTMD_INPUT_CHUNK_TYPE_TEXT) {126 LOG_ERR("failed to decode chunk: input chunk not of image/audio type\n");127 return -1;128 }129 130 const llama_model * model = llama_get_model(lctx);131 int n_mmproj_embd = llama_model_n_embd_inp(model);132 int n_pos_per_embd = mtmd_decode_use_mrope(ctx) ? 4 : 1;133 134 int32_t n_tokens = mtmd_input_chunk_get_n_tokens(chunk);135 int32_t i_batch = 0;136 int32_t n_img_batches = (n_tokens + n_batch - 1) / n_batch;137 decode_embd_batch batch_embd(encoded_embd, n_tokens, n_pos_per_embd, n_mmproj_embd);138 139 if (mtmd_decode_use_mrope(ctx)) {140 if (chunk_type == MTMD_INPUT_CHUNK_TYPE_IMAGE) {141 const auto image_tokens = mtmd_input_chunk_get_tokens_image(chunk);142 if (!image_tokens) {143 LOG_ERR("failed to decode chunk: image tokens are null\n");144 return -1;145 }146 const auto n_tokens = mtmd_image_tokens_get_n_tokens(image_tokens);147 std::vector<mtmd_decoder_pos> rel_pos(n_tokens);148 mtmd_helper_image_get_decoder_pos(image_tokens, n_past, rel_pos.data());149 batch_embd.set_position_mrope_2d(rel_pos, seq_id);150 } else if (chunk_type == MTMD_INPUT_CHUNK_TYPE_AUDIO) {151 batch_embd.set_position_mrope_1d(n_past, seq_id);152 } else {153 GGML_ABORT("invalid chunk type for M-RoPE");154 }155 } else {156 batch_embd.set_position_normal(n_past, seq_id);157 }158 159 const bool use_non_causal = mtmd_decode_use_non_causal(ctx, chunk);160 const scope_non_causal non_causal(lctx, use_non_causal);161 162 while (i_batch < n_img_batches) { // split into batches163 int pos_offset = i_batch*n_batch;164 int n_tokens_batch = std::min(n_batch, n_tokens - pos_offset);165 llama_batch batch_embd_view = batch_embd.get_view(pos_offset, n_tokens_batch);166 167 LOG_INF("decoding %s batch %d/%d, n_tokens_batch = %d\n", name, i_batch+1, n_img_batches, n_tokens_batch);168 169 int64_t t1 = ggml_time_ms();170 int32_t ret = llama_decode(lctx, batch_embd_view);171 if (ret != 0) {172 LOG_ERR("failed to decode %s\n", name);173 return ret;174 }175 176 if (callback != nullptr) {177 ret = callback(batch_embd_view, user_data);178 if (ret != 0) {179 LOG_ERR("post-decode callback failed\n");180 return ret;181 }182 }183 184 LOG_INF("%s decoded (batch %d/%d) in %" PRId64 " ms\n", name, i_batch+1, n_img_batches, ggml_time_ms() - t1);185 186 i_batch++;187 }188 189 n_past += mtmd_input_chunk_get_n_pos(chunk);190 *new_n_past = n_past;191 192 return 0;193}194 195int32_t mtmd_helper_eval_chunk_single(mtmd_context * ctx,196 struct llama_context * lctx,197 const mtmd_input_chunk * chunk,198 llama_pos n_past,199 llama_seq_id seq_id,200 int32_t n_batch,201 bool logits_last,202 llama_pos * new_n_past) {203 GGML_ASSERT(n_batch > 0);204 int32_t ret;205 llama_batch text_batch = llama_batch_init(n_batch, 0, 1);206 auto chunk_type = mtmd_input_chunk_get_type(chunk);207 208 if (chunk_type == MTMD_INPUT_CHUNK_TYPE_TEXT) {209 size_t n_tokens;210 const auto tokens = mtmd_input_chunk_get_tokens_text(chunk, &n_tokens);211 // LOG_INF("decoding text chunk, n_tokens = %zu\n", n_tokens);212 size_t i = 0;213 while (i < n_tokens) { // split into batches214 text_batch.n_tokens = 0; // clear the batch215 for (; i < n_tokens && text_batch.n_tokens < n_batch; i++) {216 int32_t j = text_batch.n_tokens;217 text_batch.token [j] = tokens[i];218 text_batch.pos [j] = n_past++;219 text_batch.n_seq_id[j] = 1;220 text_batch.seq_id [j][0] = seq_id;221 text_batch.logits [j] = false;222 223 text_batch.n_tokens++;224 }225 bool is_last_token = (i == n_tokens);226 if (logits_last && is_last_token) {227 text_batch.logits[text_batch.n_tokens - 1] = true;228 }229 ret = llama_decode(lctx, text_batch);230 if (ret != 0) {231 LOG_ERR("failed to decode text\n");232 llama_batch_free(text_batch);233 return ret;234 }235 *new_n_past += text_batch.n_tokens;236 }237 238 } else if (chunk_type == MTMD_INPUT_CHUNK_TYPE_IMAGE || chunk_type == MTMD_INPUT_CHUNK_TYPE_AUDIO) {239 const char * name = chunk_type == MTMD_INPUT_CHUNK_TYPE_IMAGE ? "image" : "audio";240 int64_t t0 = ggml_time_ms();241 242 LOG_INF("encoding %s slice...\n", name);243 244 ret = mtmd_encode_chunk(ctx, chunk);245 if (ret != 0) {246 LOG_ERR("failed to encode %s slice\n", name);247 llama_batch_free(text_batch);248 return ret;249 }250 251 LOG_INF("%s slice encoded in %" PRId64 " ms\n", name, ggml_time_ms() - t0);252 253 float * embd = mtmd_get_output_embd(ctx);254 ret = mtmd_helper_decode_image_chunk(ctx, lctx, chunk, embd, n_past, seq_id, n_batch, new_n_past, nullptr, nullptr);255 if (ret != 0) {256 LOG_ERR("failed to decode %s\n", name);257 llama_batch_free(text_batch);258 return ret;259 }260 } else {261 GGML_ABORT("chunk type not supported");262 }263 264 llama_batch_free(text_batch);265 return 0;266}267 268int32_t mtmd_helper_eval_chunks(mtmd_context * ctx,269 struct llama_context * lctx,270 const mtmd_input_chunks * chunks,271 llama_pos n_past,272 llama_seq_id seq_id,273 int32_t n_batch,274 bool logits_last,275 llama_pos * new_n_past) {276 size_t n_chunks = mtmd_input_chunks_size(chunks);277 if (n_chunks == 0) {278 LOG_WRN("no chunks to eval\n");279 return 0;280 }281 282 for (size_t i = 0; i < n_chunks; i++) {283 bool chunk_logits_last = (i == n_chunks - 1) && logits_last;284 auto chunk = mtmd_input_chunks_get(chunks, i);285 286 int32_t res = mtmd_helper_eval_chunk_single(ctx, lctx, chunk, n_past, seq_id, n_batch, chunk_logits_last, &n_past);287 if (res != 0) {288 LOG_ERR("failed to eval chunk %zu\n", i);289 return res;290 }291 *new_n_past = n_past;292 }293 294 return 0;295}296 297namespace audio_helpers {298 299static bool is_audio_file(const char * buf, size_t len) {300 if (len < 12) {301 return false;302 }303 304 // RIFF ref: https://en.wikipedia.org/wiki/Resource_Interchange_File_Format305 // WAV ref: https://www.mmsp.ece.mcgill.ca/Documents/AudioFormats/WAVE/WAVE.html306 bool is_wav = memcmp(buf, "RIFF", 4) == 0 && memcmp(buf + 8, "WAVE", 4) == 0;307 bool is_mp3 = len >= 3 && (308 memcmp(buf, "ID3", 3) == 0 ||309 // Check for MPEG sync word (simplified check)310 ((unsigned char)buf[0] == 0xFF && ((unsigned char)buf[1] & 0xE0) == 0xE0)311 );312 bool is_flac = memcmp(buf, "fLaC", 4) == 0;313 314 return is_wav || is_mp3 || is_flac;315}316 317// returns true if the buffer is a valid audio file318static bool decode_audio_from_buf(const unsigned char * buf_in, size_t len, int target_sampler_rate, std::vector<float> & pcmf32_mono) {319 ma_result result;320 const int channels = 1;321 ma_decoder_config decoder_config = ma_decoder_config_init(ma_format_f32, channels, target_sampler_rate);322 ma_decoder decoder;323 324 result = ma_decoder_init_memory(buf_in, len, &decoder_config, &decoder);325 if (result != MA_SUCCESS) {326 return false;327 }328 329 ma_uint64 frame_count;330 ma_uint64 frames_read;331 result = ma_decoder_get_length_in_pcm_frames(&decoder, &frame_count);332 if (result != MA_SUCCESS) {333 ma_decoder_uninit(&decoder);334 return false;335 }336 337 pcmf32_mono.resize(frame_count);338 result = ma_decoder_read_pcm_frames(&decoder, pcmf32_mono.data(), frame_count, &frames_read);339 if (result != MA_SUCCESS) {340 ma_decoder_uninit(&decoder);341 return false;342 }343 344#ifdef MTMD_AUDIO_DEBUG345 // save audio to wav file346 ma_encoder_config config = ma_encoder_config_init(ma_encoding_format_wav, ma_format_f32, 1, target_sampler_rate);347 ma_encoder encoder;348 ma_encoder_init_file("output.wav", &config, &encoder);349 ma_encoder_write_pcm_frames(&encoder, pcmf32_mono.data(), pcmf32_mono.size(), &frames_read);350 ma_encoder_uninit(&encoder);351#endif352 353 ma_decoder_uninit(&decoder);354 return true;355}356 357} // namespace audio_helpers358 359// Computes FNV-1a hash of the data360static std::string fnv_hash(const uint8_t * data, size_t len) {361 const uint64_t fnv_prime = 0x100000001b3ULL;362 uint64_t hash = 0xcbf29ce484222325ULL;363 364 for (size_t i = 0; i < len; ++i) {365 hash ^= data[i];366 hash *= fnv_prime;367 }368 return std::to_string(hash);369}370 371mtmd_helper_bitmap_wrapper mtmd_helper_bitmap_init_from_buf(mtmd_context * ctx, const unsigned char * buf, size_t len, bool placeholder) {372 // calculate the hash if needed373 std::string id;374 mtmd_bitmap * result = nullptr;375 376 if (!placeholder) {377 id = fnv_hash(buf, len);378 }379 380 if (audio_helpers::is_audio_file((const char *)buf, len)) {381 std::vector<float> pcmf32;382 const int sample_rate = mtmd_get_audio_sample_rate(ctx);383 if (sample_rate < 0) {384 LOG_ERR("This model does not support audio input\n");385 return {nullptr, nullptr};386 }387 if (!audio_helpers::decode_audio_from_buf(buf, len, sample_rate, pcmf32)) {388 LOG_ERR("Unable to read WAV audio file from buffer\n");389 return {nullptr, nullptr};390 }391 result = mtmd_bitmap_init_from_audio(pcmf32.size(), placeholder ? nullptr : pcmf32.data());392 mtmd_bitmap_set_id(result, id.empty() ? nullptr : id.c_str());393 return {result, nullptr};394 }395 396 // otherwise, we assume it's an image397 if (!result) {398 int nx, ny, nc;399 auto * data = stbi_load_from_memory(buf, len, &nx, &ny, &nc, 3);400 if (data) {401 result = mtmd_bitmap_init(nx, ny, placeholder ? nullptr : data);402 mtmd_bitmap_set_id(result, id.empty() ? nullptr : id.c_str());403 stbi_image_free(data);404 return {result, nullptr};405 }406 // otherwise, fallthrough to video decoding (if supported)407 }408 409 // last try: load as video410#ifdef MTMD_VIDEO411 if (!result) {412 auto params = mtmd_helper_video_init_params_default();413 auto video_ctx = mtmd_helper_video_init_from_buf(ctx, buf, len, params);414 if (!video_ctx) {415 LOG_ERR("%s: failed to decode buffer as either image/audio/video\n", __func__);416 return {nullptr, nullptr};417 }418 result = mtmd_bitmap_init_lazy(ctx,419 id.empty() ? nullptr : id.c_str(),420 video_ctx,421 [](size_t, void * user_data, mtmd_bitmap ** out_bitmap, char ** out_text) -> int {422 auto * vctx = static_cast<mtmd_helper_video *>(user_data);423 char * text = nullptr;424 int ret = mtmd_helper_video_read_next(vctx, out_bitmap, &text);425 *out_text = text; // heap-allocated by read_next; freed automatically by mtmd426 return ret;427 });428 return {result, video_ctx};429 }430#else431 if (!result) {432 LOG_ERR("%s: failed to decode buffer as either image or audio (video support not compiled in)\n", __func__);433 return {nullptr, nullptr};434 }435#endif436 437 // should not reach here438 return {nullptr, nullptr};439}440 441mtmd_helper_bitmap_wrapper mtmd_helper_bitmap_init_from_file(mtmd_context * ctx, const char * fname, bool placeholder) {442#ifdef _WIN32443 int wlen = MultiByteToWideChar(CP_UTF8, 0, fname, -1, NULL, 0);444 if (!wlen) {445 LOG_ERR("Unable to convert filename to UTF-16: %s\n", fname);446 return {nullptr, nullptr};447 }448 std::vector<wchar_t> wfname(wlen);449 wlen = MultiByteToWideChar(CP_UTF8, 0, fname, -1, wfname.data(), wlen);450 if (!wlen) {451 LOG_ERR("Unable to convert filename to UTF-16: %s\n", fname);452 return {nullptr, nullptr};453 }454 FILE * f = _wfopen(wfname.data(), L"rb");455#else456 FILE * f = fopen(fname, "rb");457#endif458 if (!f) {459 LOG_ERR("Unable to open file %s: %s\n", fname, strerror(errno));460 return {nullptr, nullptr};461 }462 463 std::vector<unsigned char> buf;464 465 fseek(f, 0, SEEK_END);466 long file_size = ftell(f);467 fseek(f, 0, SEEK_SET);468 if (file_size < 0) {469 LOG_ERR("Failed to get file size of %s\n", fname);470 fclose(f);471 return {nullptr, nullptr};472 }473 buf.resize(file_size);474 475 size_t n_read = fread(buf.data(), 1, file_size, f);476 fclose(f);477 if (n_read != (size_t)file_size) {478 LOG_ERR("Failed to read entire file %s", fname);479 return {nullptr, nullptr};480 }481 482 return mtmd_helper_bitmap_init_from_buf(ctx, buf.data(), buf.size(), placeholder);483}484 485bool mtmd_helper_support_video(mtmd_context * ctx) {486#ifdef MTMD_VIDEO487 return mtmd_support_vision(ctx);488#else489 GGML_UNUSED(ctx);490 return false;491#endif492}493 494//495// Video input helpers496//497 498#ifdef MTMD_VIDEO499 500struct mtmd_helper_video {501 mtmd_context * mctx;502 std::string path;503 std::vector<uint8_t> input_buf; // non-empty when initialized from buffer504 std::string ffmpeg_bin;505 std::string ffprobe_bin;506 float fps_target = 0.0f;507 mtmd_helper_video_info info = {};508 509 // RAII wrapper for managing subprocess510 struct subprocess_handle {511 struct subprocess_s proc = {};512 bool alive = false;513 std::thread feeder;514 515 subprocess_handle() = default;516 subprocess_handle(const subprocess_handle &) = delete;517 subprocess_handle & operator=(const subprocess_handle &) = delete;518 ~subprocess_handle() { stop(); }519 520 void stop() {521 if (alive) {522 subprocess_terminate(&proc);523 }524 // join before destroy: feeder holds a FILE* from subprocess_stdin;525 // subprocess_destroy closes it, so the thread must finish first526 if (feeder.joinable()) {527 feeder.join();528 }529 if (alive) {530 subprocess_destroy(&proc);531 alive = false;532 }533 }534 535 FILE * stdout_pipe() {536 return subprocess_stdout(&proc);537 }538 539 // buf is tied to lifetime of mtmd_helper_video, so it's guaranteed to outlive the feeder thread540 void start_feeder(const std::vector<uint8_t> & buf) {541 feeder = std::thread([this, &buf]() {542 FILE * f = subprocess_stdin(&proc);543 if (!f) {544 return;545 }546 fwrite(buf.data(), 1, buf.size(), f);547 fclose(f);548 proc.stdin_file = nullptr; // prevent double-close in subprocess_destroy549 });550 }551 };552 553 subprocess_handle sp;554 int32_t current_frame = 0;555 556 std::string prompt_start = "Video:";557 int32_t timestamp_interval_ms = 5000; // emit a timestamp text every N ms (0 = disabled)558 float next_timestamp_ms = 0.0f; // next elapsed-ms threshold at which to emit559 560 std::vector<uint8_t> frame_buf;561 std::string pending_text; // text queued to be returned before the next frame562 bool start_emitted = false;563 564 bool is_buf_input() const {565 return !input_buf.empty();566 }567 568 bool probe(float fps_target_arg) {569 const char * input_arg = is_buf_input() ? "pipe:0" : path.c_str();570 const char * cmd[] = {571 ffprobe_bin.c_str(),572 "-v", "quiet",573 "-show_entries", "stream=width,height,r_frame_rate,nb_frames,duration",574 "-select_streams", "v:0",575 "-of", "default=noprint_wrappers=1",576 input_arg,577 nullptr,578 };579 580 LOG_DBG("%s: launching:", __func__);581 for (size_t i = 0; cmd[i]; i++) { LOG_DBG(" %s", cmd[i]); }582 LOG_DBG("\n");583 584 subprocess_handle probe_sp;585 if (subprocess_create(cmd,586 subprocess_option_search_user_path | subprocess_option_inherit_environment,587 &probe_sp.proc) != 0) {588 LOG_ERR("%s: failed to launch ffprobe\n", __func__);589 return false;590 }591 probe_sp.alive = true;592 593 if (is_buf_input()) {594 probe_sp.start_feeder(input_buf);595 }596 597 uint32_t width = 0;598 uint32_t height = 0;599 float orig_fps = 0.0f;600 float duration = -1.0f;601 int32_t n_frames_orig = -1;602 char line[256];603 FILE * fp = probe_sp.stdout_pipe();604 605 while (fgets(line, sizeof(line), fp)) {606 char * eq = strchr(line, '=');607 if (!eq) continue;608 *eq = '\0';609 const char * key = line;610 const char * val = eq + 1;611 char * nl = (char *)strchr(val, '\n');612 if (nl) *nl = '\0';613 614 if (strcmp(key, "width") == 0) {615 width = (uint32_t)atoi(val);616 } else if (strcmp(key, "height") == 0) {617 height = (uint32_t)atoi(val);618 } else if (strcmp(key, "r_frame_rate") == 0) {619 orig_fps = parse_rational(val);620 } else if (strcmp(key, "nb_frames") == 0 && strcmp(val, "N/A") != 0) {621 n_frames_orig = atoi(val);622 } else if (strcmp(key, "duration") == 0 && strcmp(val, "N/A") != 0) {623 duration = (float)atof(val);624 }625 }626 627 probe_sp.stop();628 629 if (width == 0 || height == 0 || orig_fps <= 0.0f) {630 return false;631 }632 633 if (duration < 0.0f && n_frames_orig > 0) {634 duration = (float)n_frames_orig / orig_fps;635 }636 637 fps_target = fps_target_arg > 0.0f ? fps_target_arg : orig_fps;638 info.width = width;639 info.height = height;640 info.fps = fps_target;641 LOG_DBG("%s: %ux%u fps=%.2f duration=%.2fs n_frames=%d\n",642 __func__, width, height, fps_target, duration, info.n_frames);643 info.n_frames = duration > 0.0f ? (int32_t)(duration * fps_target + 0.5f) : -1;644 frame_buf.resize((size_t)width * height * 3);645 return true;646 }647 648 bool start_ffmpeg(float seek_seconds) {649 char seek_buf[64];650 char fps_buf[64];651 652 std::vector<const char *> cmd;653 cmd.push_back(ffmpeg_bin.c_str());654 655 if (!is_buf_input() && seek_seconds > 0.0f) {656 // input-side seek: fast, keyframe-accurate; only valid for seekable file inputs657 snprintf(seek_buf, sizeof(seek_buf), "%.6f", seek_seconds);658 cmd.push_back("-ss");659 cmd.push_back(seek_buf);660 }661 662 cmd.push_back("-nostdin");663 cmd.push_back("-i");664 // cache:pipe:0 wraps stdin with a seekable in-memory cache, letting ffmpeg seek665 // backwards for container headers (e.g. MP4 moov atom at end of file)666 cmd.push_back(is_buf_input() ? "cache:pipe:0" : path.c_str());667 668 if (seek_seconds > 0.0f && is_buf_input()) {669 // output-side seek: frame-accurate but decodes and discards frames up to seek point670 snprintf(seek_buf, sizeof(seek_buf), "%.6f", seek_seconds);671 cmd.push_back("-ss");672 cmd.push_back(seek_buf);673 }674 675 if (fps_target > 0.0f) {676 snprintf(fps_buf, sizeof(fps_buf), "fps=%.6f", fps_target);677 cmd.push_back("-vf");678 cmd.push_back(fps_buf);679 }680 681 cmd.push_back("-f");682 cmd.push_back("rawvideo");683 cmd.push_back("-pix_fmt");684 cmd.push_back("rgb24");685 cmd.push_back("pipe:1");686 cmd.push_back("-loglevel");687 cmd.push_back("error");688 cmd.push_back(nullptr);689 690 LOG_DBG("%s: launching:", __func__);691 for (size_t i = 0; cmd[i]; i++) {692 LOG_DBG(" %s", cmd[i]);693 }694 LOG_DBG("\n");695 696 int ret = subprocess_create(697 cmd.data(),698 subprocess_option_search_user_path | subprocess_option_inherit_environment,699 &sp.proc);700 701 sp.alive = (ret == 0);702 LOG_DBG("%s: subprocess_create ret=%d proc_alive=%d\n", __func__, ret, (int)sp.alive);703 704 if (sp.alive && is_buf_input()) {705 LOG_DBG("%s: starting feeder thread for %zu-byte buffer\n", __func__, input_buf.size());706 sp.start_feeder(input_buf);707 }708 709 return sp.alive;710 }711 712 void stop_ffmpeg() {713 sp.stop();714 }715 716 mtmd_bitmap * read_next_frame() {717 if (!sp.alive) return nullptr;718 719 FILE * fp = sp.stdout_pipe();720 const size_t frame_size = (size_t)info.width * info.height * 3;721 LOG_DBG("%s: reading frame %d, expecting %zu bytes (%ux%u)\n",722 __func__, current_frame, frame_size, info.width, info.height);723 724 size_t total_read = 0;725 while (total_read < frame_size) {726 size_t n = fread(frame_buf.data() + total_read, 1, frame_size - total_read, fp);727 if (n == 0) {728 // clean EOF only if no bytes read yet; partial frame is an error729 LOG_DBG("%s: fread returned 0 after %zu/%zu bytes (ferror=%d)\n",730 __func__, total_read, frame_size, ferror(fp));731 sp.alive = false;732 return nullptr;733 }734 total_read += n;735 }736 737 LOG_DBG("%s: frame %d read OK\n", __func__, current_frame);738 current_frame++;739 return mtmd_bitmap_init(info.width, info.height, frame_buf.data());740 }741 742 int32_t read_next(mtmd_bitmap ** out_bitmap, char ** out_text) {743 *out_bitmap = nullptr;744 *out_text = nullptr;745 746 if (!pending_text.empty()) {747 *out_text = strdup(pending_text.c_str());748 pending_text.clear();749 return *out_text ? 0 : -2;750 }751 752 LOG_DBG("%s: proc_alive=%d start_emitted=%d current_frame=%d\n",753 __func__, (int)sp.alive, (int)start_emitted, current_frame);754 755 if (!sp.alive) {756 return (current_frame == 0) ? -2 : -1;757 }758 759 if (!start_emitted) {760 start_emitted = true;761 if (!prompt_start.empty()) {762 *out_text = strdup(prompt_start.c_str());763 return *out_text ? 0 : -2;764 }765 }766 767 mtmd_bitmap * frame = read_next_frame();768 if (!frame) return -1;769 *out_bitmap = frame;770 771 if (timestamp_interval_ms > 0) {772 // current_frame was already incremented by read_next_frame(); undo for elapsed calc773 float elapsed_ms = (float)(current_frame - 1) / info.fps * 1000.0f;774 if (elapsed_ms >= next_timestamp_ms) {775 char ts_buf[32];776 float elapsed_s = elapsed_ms / 1000.0f;777 int minutes = (int)(elapsed_s / 60);778 float seconds = elapsed_s - minutes * 60.0f;779 snprintf(ts_buf, sizeof(ts_buf), "[%dm%.2fs]", minutes, seconds);780 pending_text = ts_buf;781 next_timestamp_ms += (float)timestamp_interval_ms;782 }783 }784 785 return 0;786 }787 788 static float parse_rational(const char * s) {789 int num = 0, den = 1;790 if (sscanf(s, "%d/%d", &num, &den) == 2 && den > 0) {791 return (float)num / (float)den;792 }793 float val;794 if (sscanf(s, "%f", &val) == 1) {795 return val;796 }797 return 0.0f;798 }799};800#endif801 802mtmd_helper_video_init_params mtmd_helper_video_init_params_default() {803 return {804 /* fps_target */ 4.0f,805 /* ffmpeg_bin_dir */ nullptr,806 /* timestamp_interval_ms */ 5000,807 };808}809 810static std::string video_resolve_bin(const char * bin_dir, const char * name) {811 if (!bin_dir || bin_dir[0] == '\0') {812 return name; // rely on PATH813 }814 std::string result = bin_dir;815 char last = result.back();816 if (last != '/' && last != '\\') {817#ifdef _WIN32818 result += '\\';819#else820 result += '/';821#endif822 }823 result += name;824#ifdef _WIN32825 result += ".exe";826#endif827 return result;828}829 830mtmd_helper_video * mtmd_helper_video_init(831 mtmd_context * mctx,832 const char * path,833 mtmd_helper_video_init_params params) {834#ifdef MTMD_VIDEO835 auto * ctx = new mtmd_helper_video();836 837 ctx->mctx = mctx;838 ctx->path = path;839 ctx->ffmpeg_bin = video_resolve_bin(params.ffmpeg_bin_dir, "ffmpeg");840 ctx->ffprobe_bin = video_resolve_bin(params.ffmpeg_bin_dir, "ffprobe");841 ctx->timestamp_interval_ms = params.timestamp_interval_ms;842 843 if (!ctx->probe(params.fps_target)) {844 LOG_ERR("%s: ffprobe failed for '%s' (is ffprobe in PATH?)\n", __func__, path);845 delete ctx;846 return nullptr;847 }848 849 if (!ctx->start_ffmpeg(0.0f)) {850 LOG_ERR("%s: failed to start ffmpeg for '%s' (is ffmpeg in PATH?)\n", __func__, path);851 delete ctx;852 return nullptr;853 }854 855 return ctx;856#else857 GGML_UNUSED(mctx);858 GGML_UNUSED(path);859 GGML_UNUSED(params);860 LOG_ERR("%s: video is not supported in this build (MTMD_VIDEO is set to OFF)\n", __func__);861 return nullptr;862#endif863}864 865mtmd_helper_video * mtmd_helper_video_init_from_buf(866 mtmd_context * mctx,867 const unsigned char * buf, size_t len,868 mtmd_helper_video_init_params params) {869#ifdef MTMD_VIDEO870 auto * ctx = new mtmd_helper_video();871 872 ctx->mctx = mctx;873 ctx->input_buf.assign(buf, buf + len);874 ctx->ffmpeg_bin = video_resolve_bin(params.ffmpeg_bin_dir, "ffmpeg");875 ctx->ffprobe_bin = video_resolve_bin(params.ffmpeg_bin_dir, "ffprobe");876 ctx->timestamp_interval_ms = params.timestamp_interval_ms;877 878 if (!ctx->probe(params.fps_target)) {879 LOG_ERR("%s: ffprobe failed on buffer (is ffprobe in PATH?)\n", __func__);880 delete ctx;881 return nullptr;882 }883 884 if (!ctx->start_ffmpeg(0.0f)) {885 LOG_ERR("%s: failed to start ffmpeg on buffer (is ffmpeg in PATH?)\n", __func__);886 delete ctx;887 return nullptr;888 }889 890 return ctx;891#else892 GGML_UNUSED(mctx);893 GGML_UNUSED(buf);894 GGML_UNUSED(len);895 GGML_UNUSED(params);896 LOG_ERR("%s: video is not supported in this build (MTMD_VIDEO is set to OFF)\n", __func__);897 return nullptr;898#endif899}900 901void mtmd_helper_video_free(mtmd_helper_video * ctx) {902#ifdef MTMD_VIDEO903 if (!ctx) return;904 ctx->stop_ffmpeg();905 delete ctx;906#else907 GGML_UNUSED(ctx);908 LOG_ERR("%s: video is not supported in this build (MTMD_VIDEO is set to OFF)\n", __func__);909#endif910}911 912mtmd_helper_video_info mtmd_helper_video_get_info(const mtmd_helper_video * ctx) {913#ifdef MTMD_VIDEO914 return ctx->info;915#else916 GGML_UNUSED(ctx);917 GGML_ASSERT(false && "video is not supported in this build (MTMD_VIDEO is set to OFF)");918#endif919}920 921int32_t mtmd_helper_video_read_next(mtmd_helper_video * ctx,922 mtmd_bitmap ** out_bitmap, char ** out_text) {923#ifdef MTMD_VIDEO924 if (!ctx) return -2;925 return ctx->read_next(out_bitmap, out_text);926#else927 GGML_UNUSED(ctx);928 GGML_UNUSED(out_bitmap);929 GGML_UNUSED(out_text);930 GGML_ASSERT(false && "video is not supported in this build (MTMD_VIDEO is set to OFF)");931#endif932}933 934bool mtmd_helper_model_can_chat(llama_context * lctx, mtmd_context * mctx) {935 if (!mctx) {936 return true;937 }938 939 auto * model = llama_get_model(lctx);940 auto * tmpl = llama_model_chat_template(model, nullptr);941 auto info = mtmd_gen_audio_get_info(mctx);942 943 // tts-only model cannot be used for chat (no chat template)944 bool is_tts_only = info.type != MTMD_GEN_AUDIO_TYPE_NONE && tmpl == nullptr;945 946 return !is_tts_only;947}948 