Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#include "mtmd-audio.h"2 3#define _USE_MATH_DEFINES // for M_PI4#include <cmath>5#include <cstdint>6#include <cstring>7#include <thread>8#include <vector>9#include <fstream>10#include <algorithm>11#include <functional>12 13// some of the code here is copied from whisper.cpp14 15constexpr bool DEBUG = false;16 17void mtmd_audio_cache::fill_sin_cos_table(uint32_t n) {18 sin_vals.resize(n);19 cos_vals.resize(n);20 for (uint32_t i = 0; i < n; i++) {21 double theta = (2 * M_PI * i) / n;22 sin_vals[i] = sinf(theta);23 cos_vals[i] = cosf(theta);24 }25}26 27void mtmd_audio_cache::fill_hann_window(uint32_t length, bool periodic) {28 hann_window.resize(length);29 int offset = periodic ? 0 : -1;30 for (uint32_t i = 0; i < length; i++) {31 hann_window[i] = 0.5 * (1.0 - cosf((2.0 * M_PI * i) / (length + offset)));32 }33}34 35void mtmd_audio_cache::fill_mel_filterbank_matrix(int64_t n_mel,36 int64_t n_fft,37 int sample_rate,38 float fmin,39 float fmax,40 bool slaney_area_norm,41 float scale,42 bool use_htk) {43 GGML_ASSERT(n_mel > 0 && n_fft > 1);44 if (fmax <= 0.0f) {45 fmax = 0.5f * sample_rate;46 }47 48 std::function<double(double)> hz_to_mel;49 std::function<double(double)> mel_to_hz;50 51 if (use_htk) {52 hz_to_mel = [](const double f_hz) -> double {53 return 2595.0 * log10(1.0 + f_hz / 700.0);54 };55 mel_to_hz = [](const double m) -> double {56 return 700.0 * (pow(10.0, m / 2595.0) - 1.0);57 };58 } else {59 // Slaney scale (matches librosa default)60 const double min_log_hz = 1000.0;61 const double lin_slope = 3 / 200.;62 const double min_log_mel = min_log_hz * lin_slope;63 const double log_step = log(6.4) / 27.0;64 hz_to_mel = [min_log_hz, lin_slope, log_step, min_log_mel](const double f_hz) -> double {65 return (f_hz < min_log_hz) ? f_hz * lin_slope : min_log_mel + log(f_hz / min_log_hz) / log_step;66 };67 mel_to_hz = [min_log_hz, lin_slope, log_step, min_log_mel](const double m) -> double {68 return (m < min_log_mel) ? m / lin_slope : min_log_hz * exp((m - min_log_mel) * log_step);69 };70 }71 72 // infer N_fft from n_fft_bins73 const double bin_hz_step = double(sample_rate) / double(n_fft);74 75 // mel grid: n_mel + 2 edges76 const double m_lo = hz_to_mel(fmin);77 const double m_hi = hz_to_mel(fmax);78 std::vector<double> mel_pts(n_mel + 2);79 for (int i = 0; i < n_mel + 2; ++i) {80 mel_pts[i] = m_lo + (m_hi - m_lo) * (double(i) / (n_mel + 1));81 }82 83 // convert to Hz84 std::vector<double> hz_pts(n_mel + 2);85 for (int i = 0; i < n_mel + 2; ++i) {86 hz_pts[i] = mel_to_hz(mel_pts[i]);87 }88 89 const int64_t n_fft_bins = n_fft / 2 + 1;90 91 // Validate allocation size92 if ((size_t)n_mel * (size_t)n_fft_bins > SIZE_MAX) {93 GGML_ASSERT(false && "mel filterbank allocation too large");94 }95 96 // filterbank97 std::vector<float> out((size_t)n_mel * (size_t)n_fft_bins, 0);98 for (int64_t m = 0; m < n_mel; ++m) {99 const double f_left = hz_pts[m];100 const double f_center = hz_pts[m + 1];101 const double f_right = hz_pts[m + 2];102 103 const double denom_l = std::max(1e-30, f_center - f_left);104 const double denom_r = std::max(1e-30, f_right - f_center);105 const double enorm = slaney_area_norm ? (2.0 / std::max(1e-30, f_right - f_left)) : 1.0;106 107 for (int k = 0; k < n_fft_bins; ++k) {108 const double f = k * bin_hz_step;109 double w = 0.0;110 if (f >= f_left && f <= f_center) {111 w = (f - f_left) / denom_l;112 } else if (f > f_center && f <= f_right) {113 w = (f_right - f) / denom_r;114 }115 out[size_t(m) * size_t(n_fft_bins) + size_t(k)] = float(w * enorm * scale);116 }117 }118 119 filters.n_mel = n_mel;120 filters.n_fft = n_fft;121 filters.data = std::move(out);122 123 if (DEBUG) { // debug124 for (size_t i = 0; i < filters.data.size(); ++i) {125 if (filters.data[i] != 0.0f) {126 printf("filters[%zu] = %f\n", i, filters.data[i] * 1000.0f);127 }128 }129 }130}131 132// Unified DFT implementation for both forward and inverse transforms133// Template parameters:134// Inverse: false = DFT with exp(-2πi·k·n/N), no scaling135// true = IDFT with exp(+2πi·k·n/N), scales by 1/N136// RealInput: true = input is real-valued (stride 1), avoids imaginary computations137// false = input is complex-valued (interleaved real/imag, stride 2)138template <bool Inverse, bool RealInput>139static void dft_impl(const mtmd_audio_cache & cache, const float * in, int N, float * out) {140 const int n_sin_cos_vals = cache.sin_vals.size();141 const int sin_cos_step = n_sin_cos_vals / N;142 143 constexpr float sign = Inverse ? 1.0f : -1.0f;144 const float scale = Inverse ? (1.0f / N) : 1.0f;145 146 for (int k = 0; k < N; k++) {147 float re = 0;148 float im = 0;149 150 for (int n = 0; n < N; n++) {151 int idx = (k * n * sin_cos_step) % n_sin_cos_vals;152 float cos_val = cache.cos_vals[idx];153 float sin_val = cache.sin_vals[idx];154 155 if constexpr (RealInput) {156 // Real input: in_im = 0, simplifies to:157 // re += in_re * cos_val158 // im += sign * in_re * sin_val159 float in_re = in[n];160 re += in_re * cos_val;161 im += sign * in_re * sin_val;162 } else {163 float in_re = in[n * 2 + 0];164 float in_im = in[n * 2 + 1];165 // (a + bi) * (cos + sign*i*sin) = (a*cos - sign*b*sin) + (sign*a*sin + b*cos)i166 re += in_re * cos_val - sign * in_im * sin_val;167 im += sign * in_re * sin_val + in_im * cos_val;168 }169 }170 171 out[k * 2 + 0] = re * scale;172 out[k * 2 + 1] = im * scale;173 }174}175 176// Cooley-Tukey FFT/IFFT unified implementation177// Template parameters:178// Inverse: false = FFT with exp(-2πi·k/N), no scaling179// true = IFFT with exp(+2πi·k/N), scales by 0.5 at each level180// RealInput: true = input is real-valued (stride 1)181// false = input is complex-valued (interleaved real/imag, stride 2)182template <bool Inverse, bool RealInput>183static void fft_impl(const mtmd_audio_cache & cache, float * in, int N, float * out) {184 GGML_ASSERT(N > 0);185 const int n_sin_cos_vals = cache.sin_vals.size();186 187 if (N == 1) {188 out[0] = in[0];189 if constexpr (RealInput) {190 out[1] = 0.0f;191 } else {192 out[1] = in[1];193 }194 return;195 }196 197 const int half_N = N / 2;198 if (N - half_N * 2 == 1) {199 // Odd N: fall back to DFT200 dft_impl<Inverse, RealInput>(cache, in, N, out);201 return;202 }203 204 // Split into even and odd205 if constexpr (RealInput) {206 // Real input: stride is 1, copy only real values207 float * even = in + N;208 for (int i = 0; i < half_N; ++i) {209 even[i] = in[2 * i];210 }211 float * even_fft = out + 2 * N;212 fft_impl<Inverse, true>(cache, even, half_N, even_fft);213 214 float * odd = even;215 for (int i = 0; i < half_N; ++i) {216 odd[i] = in[2 * i + 1];217 }218 float * odd_fft = even_fft + N;219 fft_impl<Inverse, true>(cache, odd, half_N, odd_fft);220 } else {221 // Complex input: stride is 2, copy complex pairs222 float * even = in + N * 2;223 for (int i = 0; i < half_N; ++i) {224 even[i * 2 + 0] = in[2 * i * 2 + 0];225 even[i * 2 + 1] = in[2 * i * 2 + 1];226 }227 float * even_fft = out + 2 * N;228 fft_impl<Inverse, false>(cache, even, half_N, even_fft);229 230 float * odd = even;231 for (int i = 0; i < half_N; ++i) {232 odd[i * 2 + 0] = in[(2 * i + 1) * 2 + 0];233 odd[i * 2 + 1] = in[(2 * i + 1) * 2 + 1];234 }235 float * odd_fft = even_fft + N;236 fft_impl<Inverse, false>(cache, odd, half_N, odd_fft);237 }238 239 float * even_fft = out + 2 * N;240 float * odd_fft = even_fft + N;241 242 const int sin_cos_step = n_sin_cos_vals / N;243 244 constexpr float sign = Inverse ? 1.0f : -1.0f;245 constexpr float scale = Inverse ? 0.5f : 1.0f;246 247 for (int k = 0; k < half_N; k++) {248 int idx = k * sin_cos_step; // t = 2*M_PI*k/N249 float re = cache.cos_vals[idx];250 float im = sign * cache.sin_vals[idx];251 252 float re_odd = odd_fft[2 * k + 0];253 float im_odd = odd_fft[2 * k + 1];254 255 out[2 * k + 0] = scale * (even_fft[2 * k + 0] + re * re_odd - im * im_odd);256 out[2 * k + 1] = scale * (even_fft[2 * k + 1] + re * im_odd + im * re_odd);257 258 out[2 * (k + half_N) + 0] = scale * (even_fft[2 * k + 0] - re * re_odd + im * im_odd);259 out[2 * (k + half_N) + 1] = scale * (even_fft[2 * k + 1] - re * im_odd - im * re_odd);260 }261}262 263// Forward FFT for real input (used by mel spectrogram)264static void fft(const mtmd_audio_cache & cache, float * in, int N, float * out) {265 fft_impl<false, true>(cache, in, N, out);266}267 268// Inverse FFT for complex input269static void ifft(const mtmd_audio_cache & cache, float * in, int N, float * out) {270 fft_impl<true, false>(cache, in, N, out);271}272 273struct filter_params {274 int64_t n_mel;275 int64_t n_fft_bins;276 int32_t hann_window_size;277 int32_t hop_length;278 int32_t sample_rate;279 bool no_padding = false;280 bool center_padding = false;281 float preemph = 0.f;282 bool use_natural_log = false;283 bool norm_per_feature = false;284 bool use_magnitude = false; // |X| instead of |X|^2285 float mel_floor = 5.960464477539063e-08f;286};287 288static void log_mel_spectrogram_worker_thread(int ith,289 const float * hann,290 const std::vector<float> & samples,291 int n_samples,292 int frame_size,293 int frame_step,294 int n_threads,295 const filter_params & params,296 const mtmd_audio_cache & cache,297 mtmd_audio_mel & out) {298 std::vector<float> fft_in(frame_size * 2, 0.0);299 std::vector<float> fft_out(frame_size * 2 * 2 * 2);300 301 int64_t n_fft_bins = params.n_fft_bins;302 int64_t i = ith;303 304 const auto & filters = cache.filters;305 306 // make sure n_fft == 1 + (WHISPER_N_FFT / 2), bin_0 to bin_nyquist307 GGML_ASSERT(n_fft_bins == 1 + (frame_size / 2));308 GGML_ASSERT(cache.sin_vals.size() == cache.cos_vals.size());309 // calculate FFT only when fft_in are not all zero310 for (; i < std::min((int64_t)(n_samples / frame_step + 1), out.n_len); i += n_threads) {311 const int64_t offset = i * frame_step;312 313 // apply Hann window (~10% faster)314 const int valid_len = std::min(frame_size, std::max(0, n_samples - (int)offset));315 for (int j = 0; j < valid_len; j++) {316 fft_in[j] = hann[j] * samples[offset + j];317 }318 319 // fill the rest with zeros320 if (valid_len < frame_size) {321 std::fill(fft_in.begin() + valid_len, fft_in.end(), 0.0);322 }323 324 // FFT325 fft(cache, fft_in.data(), frame_size, fft_out.data());326 327 // Calculate modulus^2 (power) or modulus (magnitude)328 for (int j = 0; j < n_fft_bins; j++) {329 float power = (fft_out[2 * j + 0] * fft_out[2 * j + 0] + fft_out[2 * j + 1] * fft_out[2 * j + 1]);330 fft_out[j] = params.use_magnitude ? sqrtf(power) : power;331 }332 333 // mel spectrogram334 for (int64_t j = 0; j < out.n_mel; j++) {335 double sum = 0.0;336 // unroll loop (suggested by GH user @lunixbochs)337 int k = 0;338 for (k = 0; k < n_fft_bins - 3; k += 4) {339 size_t idx = size_t(j) * size_t(n_fft_bins) + size_t(k);340 sum +=341 fft_out[k + 0] * filters.data[idx + 0] +342 fft_out[k + 1] * filters.data[idx + 1] +343 fft_out[k + 2] * filters.data[idx + 2] +344 fft_out[k + 3] * filters.data[idx + 3];345 }346 // handle n_fft remainder347 for (; k < n_fft_bins; k++) {348 sum += fft_out[k] * filters.data[(size_t)j * n_fft_bins + k];349 }350 sum = std::max(sum, (double)params.mel_floor);351 sum = params.use_natural_log352 ? log(sum)353 : log10(sum);354 out.data[(size_t)j * out.n_len + i] = sum;355 }356 }357 358 // Otherwise fft_out are all zero359 double sum = params.use_natural_log ? log(1e-10) : log10(1e-10);360 for (; i < out.n_len; i += n_threads) {361 for (int64_t j = 0; j < out.n_mel; j++) {362 out.data[(size_t)j * out.n_len + i] = sum;363 }364 }365}366 367// ref: https://github.com/openai/whisper/blob/main/whisper/audio.py#L110-L157368static bool log_mel_spectrogram(369 const float * samples,370 const int n_samples_in,371 const int n_threads,372 const filter_params & params,373 const mtmd_audio_cache & cache,374 mtmd_audio_mel & out) {375 //const int64_t t_start_us = ggml_time_us();376 377 out.n_len_org = n_samples_in;378 int n_samples = n_samples_in;379 380 // Hann window381 const float * hann = cache.hann_window.data();382 const int frame_size = (params.n_fft_bins - 1) * 2;383 const int frame_step = params.hop_length;384 385 // Padding386 std::vector<float> samples_padded;387 if (params.no_padding) {388 // no padding, use samples as-is389 samples_padded = std::vector<float>(samples, samples + n_samples);390 samples = samples_padded.data();391 n_samples = samples_padded.size();392 } else if (params.center_padding) {393 const auto pad_amount = frame_size / 2;394 samples_padded = std::vector<float>(n_samples + 2 * pad_amount, 0);395 std::copy(samples, samples + n_samples, samples_padded.data() + pad_amount);396 samples = samples_padded.data();397 n_samples = samples_padded.size();398 } else {399 // existing padding logic400 int64_t stage_1_pad = params.sample_rate * 30;401 int64_t stage_2_pad = frame_size / 2;402 samples_padded.resize(n_samples + stage_1_pad + stage_2_pad * 2);403 std::copy(samples, samples + n_samples, samples_padded.begin() + stage_2_pad);404 // pad 30 seconds of zeros at the end of audio (480,000 samples) + reflective pad 200 samples at the end of audio405 std::fill(samples_padded.begin() + n_samples + stage_2_pad, samples_padded.begin() + n_samples + stage_1_pad + 2 * stage_2_pad, 0);406 // reflective pad 200 samples at the beginning of audio407 if (n_samples < stage_2_pad + 1) {408 // TODO: Handle short audio differently or return error409 return false;410 }411 std::reverse_copy(samples + 1, samples + 1 + stage_2_pad, samples_padded.begin());412 413 // expose the padded buffer to downstream FFT and to out.n_len computation414 // mirrors the no_padding and center_padding branches above415 samples = samples_padded.data();416 n_samples = samples_padded.size();417 }418 419 // preemphasis420 if (params.preemph) {421 const int pad_amount = frame_size / 2;422 const float preemph = 0.97f;423 float prev = samples_padded[pad_amount];424 for (int i = pad_amount + 1; i + pad_amount < n_samples; ++i) {425 float cur = samples_padded[i];426 samples_padded[i] = cur - preemph * prev;427 prev = cur;428 }429 }430 431 // pad hann window if it's smaller than frame_size432 // TODO: probably unnecessary here? (or better doing it in g_cache?)433 std::vector<float> hann_window_padded;434 if (params.hann_window_size < frame_size) {435 hann_window_padded.resize(frame_size);436 const int padding = (frame_size - params.hann_window_size) / 2;437 std::copy(hann, hann + params.hann_window_size, &hann_window_padded[padding]);438 hann = hann_window_padded.data();439 }440 441 442 GGML_ASSERT(params.n_fft_bins > 0);443 GGML_ASSERT(params.hop_length > 0);444 out.n_mel = params.n_mel;445 out.n_len = (n_samples - frame_size) / frame_step + 1;446 // Validate dimensions before allocation to prevent integer overflow447 if (out.n_mel <= 0 || out.n_len <= 0) {448 LOG_ERR("%s: invalid mel dimensions n_mel=%lld n_len=%lld\n", __func__, (long long)out.n_mel, (long long)out.n_len);449 return false;450 }451 const size_t total_size = (size_t)out.n_mel * (size_t)out.n_len;452 if (total_size > SIZE_MAX / sizeof(float)) {453 LOG_ERR("%s: size overflow: n_mel=%lld n_len=%lld\n", __func__, (long long)out.n_mel, (long long)out.n_len);454 return false;455 }456 if (n_samples < frame_size) {457 LOG_ERR("%s: not enough samples after padding\n", __func__);458 return false;459 }460 out.data.resize(total_size);461 462 {463 std::vector<std::thread> workers(n_threads - 1);464 for (int iw = 0; iw < n_threads - 1; ++iw) {465 workers[iw] =466 std::thread(log_mel_spectrogram_worker_thread, iw + 1, hann, std::cref(samples_padded), n_samples,467 frame_size, frame_step, n_threads, std::cref(params), std::cref(cache), std::ref(out));468 }469 470 // main thread471 log_mel_spectrogram_worker_thread(0, hann, samples_padded, n_samples, frame_size, frame_step, n_threads, params,472 cache, out);473 for (int iw = 0; iw < n_threads - 1; ++iw) {474 workers[iw].join();475 }476 }477 478 const int64_t effective_n_len = n_samples_in / frame_step;479 if (params.norm_per_feature) {480 GGML_ASSERT(effective_n_len > 1);481 for (int64_t i = 0; i < out.n_mel; i++) {482 double mean = 0;483 for (int64_t j = 0; j < effective_n_len; ++j) {484 mean += out.data[(size_t)i * out.n_len + j];485 }486 mean /= effective_n_len;487 488 double var = 0.0;489 for (int64_t j = 0; j < effective_n_len; ++j) {490 const double value = out.data[(size_t)i * out.n_len + j] - mean;491 var += value * value;492 }493 var /= effective_n_len - 1; // unbiased494 const double mstd = std::sqrt(var + 1e-5);495 496 for (int64_t j = 0; j < effective_n_len; ++j) {497 auto &value = out.data[(size_t)i * out.n_len + j];498 value = (value - mean) / mstd;499 }500 501 // pad the rest with zeros502 for (int64_t j = effective_n_len; j < out.n_len; ++j) {503 out.data[(size_t)i * out.n_len + j] = 0.0;504 }505 }506 } else if (!params.no_padding) {507 // Whisper-style clamping and normalization (NOT used by Gemma4)508 double mmax = -1e20;509 const size_t mel_size = (size_t)out.n_mel * (size_t)out.n_len;510 for (size_t i = 0; i < mel_size; i++) {511 if (out.data[i] > mmax) {512 mmax = out.data[i];513 }514 }515 516 mmax -= 8.0;517 518 for (size_t i = 0; i < mel_size; i++) {519 if (out.data[i] < mmax) {520 out.data[i] = mmax;521 }522 out.data[i] = (out.data[i] + 4.0)/4.0;523 }524 }525 526 // Dump log_mel_spectrogram527 if (DEBUG) {528 std::ofstream outFile("log_mel_spectrogram.json");529 outFile << "[";530 for (uint64_t i = 0; i < out.data.size() - 1; i++) {531 outFile << out.data[i] << ", ";532 }533 outFile << out.data[out.data.size() - 1] << "]";534 outFile.close();535 }536 537 return true;538}539 540//541// mtmd_audio_preprocessor_whisper542//543 544void mtmd_audio_preprocessor_whisper::initialize() {545 cache.fill_sin_cos_table(hparams.audio_n_fft);546 cache.fill_hann_window(hparams.audio_window_len, true);547 cache.fill_mel_filterbank_matrix(hparams.n_mel_bins, hparams.audio_n_fft, hparams.audio_sample_rate);548}549 550bool mtmd_audio_preprocessor_whisper::preprocess(const float * samples,551 size_t n_samples,552 std::vector<mtmd_audio_mel> & output) {553 if (n_samples == 0) {554 // empty audio555 return false;556 }557 558 std::vector<float> smpl;559 // reflection padding needs one sample plus half an FFT window560 size_t min_samples = (size_t) hparams.audio_n_fft / 2 + 1;561 if (n_samples < min_samples) {562 smpl.resize(min_samples, 0.0f);563 std::memcpy(smpl.data(), samples, n_samples * sizeof(float));564 samples = smpl.data();565 n_samples = smpl.size();566 }567 568 filter_params params;569 params.n_mel = hparams.n_mel_bins;570 params.n_fft_bins = 1 + (hparams.audio_n_fft / 2);571 params.hann_window_size = hparams.audio_window_len;572 params.hop_length = hparams.audio_hop_len;573 params.sample_rate = hparams.audio_sample_rate;574 params.center_padding = false;575 params.preemph = 0.0f; // disabled576 params.use_natural_log = false;577 params.norm_per_feature = false;578 579 // make sure the cache is initialized580 GGML_ASSERT(!cache.sin_vals.empty());581 GGML_ASSERT(!cache.cos_vals.empty());582 GGML_ASSERT(!cache.filters.data.empty());583 584 mtmd_audio_mel out_full;585 bool ok = log_mel_spectrogram(samples, n_samples,586 4, // n_threads587 params, cache, out_full);588 if (!ok) {589 return false;590 }591 592 // because the cgraph in clip.cpp only accepts 3000 frames each, we need to split the mel593 // we always expect the mel to have 3000 silent frames at the end594 if (DEBUG) {595 printf("output: n_mel = %d, n_len = %d\n", (int) out_full.n_mel, (int) out_full.n_len);596 }597 const size_t frames_per_chunk = 3000;598 GGML_ASSERT((size_t) out_full.n_len > frames_per_chunk);599 for (size_t off = 0; off < (size_t) out_full.n_len; off += frames_per_chunk) {600 int64_t n_len = std::min((int64_t)frames_per_chunk, out_full.n_len - (int64_t)off);601 if (n_len < (int64_t)frames_per_chunk) {602 break; // last incomplete chunk will always be a padded chunk, safe to ignore603 }604 605 mtmd_audio_mel out_chunk;606 out_chunk.n_len = n_len;607 out_chunk.n_mel = out_full.n_mel;608 out_chunk.n_len_org = out_full.n_mel; // unused609 out_chunk.data.reserve((size_t)out_chunk.n_mel * (size_t)out_chunk.n_len);610 611 for (int64_t i = 0; i < out_full.n_mel; i++) {612 auto src = out_full.data.begin() + (size_t)i * out_full.n_len + off;613 out_chunk.data.insert(out_chunk.data.end(), src, src + frames_per_chunk);614 }615 616 output.push_back(std::move(out_chunk));617 }618 619 return true;620}621 622//623// mtmd_audio_preprocessor_qwen3a624//625// Matches the Python WhisperFeatureExtractor called with truncation=False:626// - reflection padding of n_fft/2 samples at each end (center=True)627// - Whisper-style log10 + (max-8)/4 normalization applied to full audio628// - output split into ≤30s (3000 mel frames) windows, each padded to a629// multiple of 200 frames (n_window * 2) for the cgraph batch view630//631 632void mtmd_audio_preprocessor_qwen3a::initialize() {633 cache.fill_sin_cos_table(hparams.audio_n_fft);634 cache.fill_hann_window(hparams.audio_window_len, true);635 cache.fill_mel_filterbank_matrix(hparams.n_mel_bins, hparams.audio_n_fft, hparams.audio_sample_rate);636}637 638bool mtmd_audio_preprocessor_qwen3a::preprocess(const float * samples,639 size_t n_samples,640 std::vector<mtmd_audio_mel> & output) {641 if (n_samples == 0) {642 return false;643 }644 645 GGML_ASSERT(!cache.sin_vals.empty());646 GGML_ASSERT(!cache.cos_vals.empty());647 GGML_ASSERT(!cache.filters.data.empty());648 649 // Reflection-pad n_fft/2 samples at each end, matching WhisperFeatureExtractor center=True650 const int pad = hparams.audio_n_fft / 2; // = 200651 652 std::vector<float> padded(n_samples + 2 * pad, 0.0f);653 // Reflect start: padded[0..pad-1] = samples[pad..1] (reversed)654 for (int i = 0; i < pad; i++) {655 int src = pad - i; // samples[pad], samples[pad-1], ..., samples[1]656 padded[i] = (src < (int)n_samples) ? samples[src] : 0.0f;657 }658 std::copy(samples, samples + n_samples, padded.begin() + pad);659 // Reflect end: padded[n+pad..n+2*pad-1] = samples[n-2..n-pad-1] (reversed)660 for (int i = 0; i < pad; i++) {661 int src = (int)n_samples - 2 - i; // samples[n-2], samples[n-3], ...662 padded[n_samples + pad + i] = (src >= 0) ? samples[src] : 0.0f;663 }664 665 filter_params params;666 params.n_mel = hparams.n_mel_bins;667 params.n_fft_bins = 1 + (hparams.audio_n_fft / 2);668 params.hann_window_size = hparams.audio_window_len;669 params.hop_length = hparams.audio_hop_len;670 params.sample_rate = hparams.audio_sample_rate;671 params.no_padding = true; // reflection padding already applied above672 params.use_natural_log = false; // log10673 674 mtmd_audio_mel mel_full;675 bool ok = log_mel_spectrogram(padded.data(), (int)padded.size(), 4, params, cache, mel_full);676 if (!ok) {677 return false;678 }679 680 // Whisper-style normalization: clamp to (max - 8), scale to [-1, 1]681 {682 double mmax = -1e20;683 for (float v : mel_full.data) {684 if (v > mmax) mmax = v;685 }686 mmax -= 8.0;687 for (float & v : mel_full.data) {688 v = (std::max((double)v, mmax) + 4.0) / 4.0;689 }690 }691 692 // The effective frame count: center-padded STFT gives ~n_samples/hop_length frames.693 // We take min(mel_full.n_len, n_samples/hop + 1) to avoid including excess frames.694 const int64_t n_eff = std::min(mel_full.n_len,695 (int64_t)(n_samples / hparams.audio_hop_len) + 1);696 697 // Split into inference windows matching n_window_infer=800 from model config.698 // Each window is padded to the next multiple of chunk_size for the cgraph.699 // The mtmd caller loops over output entries, so long audio is handled automatically.700 const int chunk_size = 100; // conv sub-chunk size (n_window * 2, n_window=50)701 const int window_size = 800; // mel frames per forward pass (n_window_infer=800)702 703 for (int64_t off = 0; off < n_eff; off += window_size) {704 const int64_t win_eff = std::min((int64_t)window_size, n_eff - off);705 const int64_t n_chunks = (win_eff + chunk_size - 1) / chunk_size;706 const int64_t n_padded = n_chunks * chunk_size;707 708 mtmd_audio_mel out;709 out.n_mel = mel_full.n_mel;710 out.n_len = n_padded;711 out.n_len_org = win_eff;712 out.data.assign((size_t)out.n_mel * (size_t)out.n_len, 0.0f);713 for (int64_t m = 0; m < out.n_mel; m++) {714 const int64_t copy_len = std::min((int64_t)win_eff, mel_full.n_len - off);715 if (copy_len > 0) {716 std::copy(mel_full.data.begin() + (size_t)m * mel_full.n_len + off,717 mel_full.data.begin() + (size_t)m * mel_full.n_len + off + copy_len,718 out.data.begin() + (size_t)m * out.n_len);719 }720 }721 output.push_back(std::move(out));722 }723 return true;724}725 726//727// mtmd_audio_preprocessor_mimo_audio728//729// Matches torchaudio.transforms.MelSpectrogram(power=1.0, center=True) followed by730// log(clip(spec, min=1e-7)): HTK mel scale, no Slaney area norm, magnitude (not power)731// spectrogram, natural log, reflect-padded by n_fft/2 on each side.732//733 734void mtmd_audio_preprocessor_mimo_audio::initialize() {735 cache.fill_sin_cos_table(hparams.audio_n_fft);736 cache.fill_hann_window(hparams.audio_window_len, true);737 cache.fill_mel_filterbank_matrix(738 hparams.n_mel_bins, hparams.audio_n_fft, hparams.audio_sample_rate,739 0.0f, hparams.audio_sample_rate / 2.0f,740 /*slaney_area_norm=*/ false,741 /*scale=*/ 1.0f,742 /*use_htk=*/ true743 );744}745 746bool mtmd_audio_preprocessor_mimo_audio::preprocess(const float * samples,747 size_t n_samples,748 std::vector<mtmd_audio_mel> & output) {749 if (n_samples == 0) {750 return false;751 }752 753 GGML_ASSERT(!cache.sin_vals.empty());754 GGML_ASSERT(!cache.cos_vals.empty());755 GGML_ASSERT(!cache.filters.data.empty());756 757 const int pad = hparams.audio_n_fft / 2;758 759 std::vector<float> padded(n_samples + 2 * pad, 0.0f);760 for (int i = 0; i < pad; i++) {761 int src = pad - i;762 padded[i] = (src < (int)n_samples) ? samples[src] : 0.0f;763 }764 std::copy(samples, samples + n_samples, padded.begin() + pad);765 for (int i = 0; i < pad; i++) {766 int src = (int)n_samples - 2 - i;767 padded[n_samples + pad + i] = (src >= 0) ? samples[src] : 0.0f;768 }769 770 filter_params params;771 params.n_mel = hparams.n_mel_bins;772 params.n_fft_bins = 1 + (hparams.audio_n_fft / 2);773 params.hann_window_size = hparams.audio_window_len;774 params.hop_length = hparams.audio_hop_len;775 params.sample_rate = hparams.audio_sample_rate;776 params.no_padding = true; // reflect padding already applied above777 params.use_natural_log = true;778 params.use_magnitude = true;779 params.mel_floor = 1e-7f;780 params.norm_per_feature = false;781 782 mtmd_audio_mel out;783 bool ok = log_mel_spectrogram(padded.data(), (int)padded.size(), 4, params, cache, out);784 if (!ok) {785 return false;786 }787 788 output.push_back(std::move(out));789 return true;790}791 792//793// mtmd_audio_preprocessor_qwen3tts_spk794//795// same as mel_spectrogram() in modeling_qwen3_tts.py796// ECAPA-TDNN takes the whole clip in one pass, so no Whisper-style chunking or normalization797//798 799void mtmd_audio_preprocessor_qwen3tts_spk::initialize() {800 cache.fill_sin_cos_table(hparams.audio_n_fft);801 cache.fill_hann_window(hparams.audio_window_len, true);802 cache.fill_mel_filterbank_matrix(hparams.n_mel_bins, hparams.audio_n_fft, hparams.audio_sample_rate);803}804 805bool mtmd_audio_preprocessor_qwen3tts_spk::preprocess(const float * samples,806 size_t n_samples,807 std::vector<mtmd_audio_mel> & output) {808 if (n_samples == 0) {809 return false;810 }811 812 GGML_ASSERT(!cache.sin_vals.empty());813 GGML_ASSERT(!cache.cos_vals.empty());814 GGML_ASSERT(!cache.filters.data.empty());815 816 // reflect pad by (n_fft - hop) / 2 = 384, matching center=False STFT framing817 const int pad = (hparams.audio_n_fft - hparams.audio_hop_len) / 2;818 if (n_samples < (size_t) pad + 1) {819 return false;820 }821 822 std::vector<float> padded(n_samples + 2 * pad, 0.0f);823 for (int i = 0; i < pad; i++) {824 padded[i] = samples[pad - i];825 }826 std::copy(samples, samples + n_samples, padded.begin() + pad);827 for (int i = 0; i < pad; i++) {828 padded[n_samples + pad + i] = samples[n_samples - 2 - i];829 }830 831 filter_params params;832 params.n_mel = hparams.n_mel_bins;833 params.n_fft_bins = 1 + (hparams.audio_n_fft / 2);834 params.hann_window_size = hparams.audio_window_len;835 params.hop_length = hparams.audio_hop_len;836 params.sample_rate = hparams.audio_sample_rate;837 params.no_padding = true; // reflect padding already applied above838 params.use_natural_log = true;839 params.use_magnitude = true;840 params.mel_floor = 1e-5f;841 842 mtmd_audio_mel out;843 bool ok = log_mel_spectrogram(padded.data(), (int) padded.size(), 4, params, cache, out);844 if (!ok) {845 return false;846 }847 848 output.push_back(std::move(out));849 return true;850}851 852//853// mtmd_audio_preprocessor_conformer854//855 856void mtmd_audio_preprocessor_conformer::initialize() {857 cache.fill_sin_cos_table(hparams.audio_n_fft);858 cache.fill_hann_window(hparams.audio_window_len, true);859 cache.fill_mel_filterbank_matrix(hparams.n_mel_bins, hparams.audio_n_fft, hparams.audio_sample_rate);860}861 862bool mtmd_audio_preprocessor_conformer::preprocess(const float * samples,863 size_t n_samples,864 std::vector<mtmd_audio_mel> & output) {865 // empty audio866 if (n_samples == 0) {867 return false;868 }869 870 filter_params params;871 params.n_mel = hparams.n_mel_bins;872 params.n_fft_bins = 1 + (hparams.audio_n_fft / 2);873 params.hann_window_size = hparams.audio_window_len;874 params.hop_length = hparams.audio_hop_len;875 params.sample_rate = hparams.audio_sample_rate;876 params.center_padding = true;877 params.preemph = 0.97f;878 params.use_natural_log = true;879 params.norm_per_feature = true;880 881 // make sure the cache is initialized882 GGML_ASSERT(!cache.sin_vals.empty());883 GGML_ASSERT(!cache.cos_vals.empty());884 GGML_ASSERT(!cache.filters.data.empty());885 886 mtmd_audio_mel out_full;887 bool ok = log_mel_spectrogram(samples, n_samples,888 4, // n_threads889 params, cache, out_full);890 if (!ok) {891 return false;892 }893 894 output.push_back(std::move(out_full));895 return true;896}897 898//899// mtmd_audio_preprocessor_granite_speech900//901 902void mtmd_audio_preprocessor_granite_speech::initialize() {903 cache.fill_sin_cos_table(hparams.audio_n_fft);904 cache.fill_hann_window(hparams.audio_window_len, true);905 cache.fill_mel_filterbank_matrix(906 hparams.n_mel_bins / 2, hparams.audio_n_fft, hparams.audio_sample_rate,907 0.0f, -1.0f, false, 1.0f, true);908}909 910bool mtmd_audio_preprocessor_granite_speech::preprocess(const float * samples,911 size_t n_samples,912 std::vector<mtmd_audio_mel> & output) {913 if (n_samples == 0) {914 return false;915 }916 917 GGML_ASSERT(!cache.sin_vals.empty());918 GGML_ASSERT(!cache.cos_vals.empty());919 GGML_ASSERT(!cache.filters.data.empty());920 921 const int n_fft = hparams.audio_n_fft;922 const int pad = n_fft / 2;923 924 // reflect padding925 const int n_padded = (int)n_samples + 2 * pad;926 std::vector<float> padded(n_padded, 0.0f);927 std::copy(samples, samples + n_samples, padded.data() + pad);928 for (int i = 0; i < pad; i++) {929 int src = i + 1;930 if (src >= (int)n_samples) {931 src = (int)n_samples - 1;932 }933 padded[pad - 1 - i] = samples[src];934 }935 for (int i = 0; i < pad; i++) {936 int src = (int)n_samples - 2 - i;937 if (src < 0) {938 src = 0;939 }940 padded[pad + (int)n_samples + i] = samples[src];941 }942 943 filter_params params;944 params.n_mel = hparams.n_mel_bins / 2;945 params.n_fft_bins = 1 + (n_fft / 2);946 params.hann_window_size = hparams.audio_window_len;947 params.hop_length = hparams.audio_hop_len;948 params.sample_rate = hparams.audio_sample_rate;949 params.no_padding = true;950 params.center_padding = false;951 params.preemph = 0.0f;952 params.use_natural_log = false;953 params.norm_per_feature = false;954 params.mel_floor = 1e-10f;955 956 mtmd_audio_mel mel;957 if (!log_mel_spectrogram(padded.data(), n_padded, 4, params, cache, mel)) {958 return false;959 }960 961 double mmax = -1e20;962 const size_t mel_size = (size_t)mel.n_mel * (size_t)mel.n_len;963 for (size_t i = 0; i < mel_size; i++) {964 if (mel.data[i] > mmax) {965 mmax = mel.data[i];966 }967 }968 mmax -= 8.0;969 970 for (size_t i = 0; i < mel_size; i++) {971 if (mel.data[i] < mmax) {972 mel.data[i] = mmax;973 }974 mel.data[i] = (mel.data[i] + 4.0) / 4.0;975 }976 977 int64_t n_frames = mel.n_len;978 if (n_frames % 2 == 1) {979 n_frames--;980 }981 const int64_t n_mel = mel.n_mel;982 const int64_t n_stacked = n_frames / 2;983 984 mtmd_audio_mel stacked;985 stacked.n_mel = 2 * n_mel;986 stacked.n_len = n_stacked;987 stacked.n_len_org = (int64_t)n_samples;988 stacked.data.resize((size_t)2 * (size_t)n_mel * (size_t)n_stacked);989 990 for (int64_t t = 0; t < n_stacked; t++) {991 for (int64_t m = 0; m < n_mel; m++) {992 stacked.data[(size_t)m * n_stacked + t] = mel.data[(size_t)m * mel.n_len + 2 * t];993 stacked.data[(size_t)(m + n_mel) * n_stacked + t] = mel.data[(size_t)m * mel.n_len + 2 * t + 1];994 }995 }996 997 output.push_back(std::move(stacked));998 return true;999}1000 1001//1002// mtmd_audio_preprocessor_gemma4a1003//1004 1005void mtmd_audio_preprocessor_gemma4a::initialize() {1006 cache.fill_sin_cos_table(hparams.audio_n_fft);1007 1008 // Standard periodic Hann window, zero-padded to FFT size1009 cache.hann_window.assign(hparams.audio_n_fft, 0.0f);1010 for (uint32_t i = 0; i < (uint32_t)hparams.audio_window_len; i++) {1011 cache.hann_window[i] = 0.5f - 0.5f * cosf((2.0f * (float)M_PI * i) / hparams.audio_window_len);1012 }1013 1014 // HTK mel scale, no Slaney area normalization1015 cache.fill_mel_filterbank_matrix(1016 hparams.n_mel_bins, hparams.audio_n_fft, hparams.audio_sample_rate,1017 0.0f, hparams.audio_sample_rate / 2.0f,1018 /*slaney_area_norm=*/ false,1019 /*scale=*/ 1.0f,1020 /*use_htk=*/ true1021 );1022}1023 1024bool mtmd_audio_preprocessor_gemma4a::preprocess(const float * samples,1025 size_t n_samples,1026 std::vector<mtmd_audio_mel> & output) {1027 if (n_samples == 0) {1028 return false;1029 }1030 1031 GGML_ASSERT(!cache.sin_vals.empty());1032 GGML_ASSERT(!cache.cos_vals.empty());1033 GGML_ASSERT(!cache.filters.data.empty());1034 1035 filter_params params;1036 params.n_mel = hparams.n_mel_bins;1037 params.n_fft_bins = 1 + (hparams.audio_n_fft / 2);1038 params.hann_window_size = hparams.audio_n_fft; // window is zero-padded to FFT size1039 params.hop_length = hparams.audio_hop_len;1040 params.sample_rate = hparams.audio_sample_rate;1041 params.no_padding = true;1042 params.center_padding = false;1043 params.preemph = 0.0f;1044 params.use_natural_log = true;1045 params.use_magnitude = true;1046 params.mel_floor = 0.001f;1047 params.norm_per_feature = false;1048 1049 // Split into 30-second chunks (model context limit, ~750 tokens each)1050 const size_t chunk_samples = 30 * hparams.audio_sample_rate;1051 for (size_t off = 0; off < n_samples; off += chunk_samples) {1052 const float * chunk_ptr = samples + off;1053 size_t chunk_len = std::min(chunk_samples, n_samples - off);1054 1055 // Semicausal left-padding + right-padding to match PyTorch frame count1056 const int pad_left = hparams.audio_window_len / 2;1057 const int fft_size = hparams.audio_n_fft;1058 const int hop = hparams.audio_hop_len;1059 const int n_with_left = (int)chunk_len + pad_left;1060 // PyTorch: unfold(size=frame_length+1, step=hop) on semicausal-padded waveform1061 const int64_t pt_frames = (n_with_left - (hparams.audio_window_len + 1)) / hop + 1;1062 const int64_t n_padded_needed = (pt_frames - 1) * hop + fft_size;1063 const int total_pad = std::max((int)(n_padded_needed - (int)chunk_len), pad_left);1064 std::vector<float> padded_samples(total_pad + chunk_len, 0.0f);1065 std::copy(chunk_ptr, chunk_ptr + chunk_len, padded_samples.data() + pad_left);1066 1067 mtmd_audio_mel out_chunk;1068 bool ok = log_mel_spectrogram(padded_samples.data(), padded_samples.size(), 4, params, cache, out_chunk);1069 if (!ok) {1070 return false;1071 }1072 1073 // Trim to PyTorch frame count1074 out_chunk.n_len = std::min(out_chunk.n_len, pt_frames);1075 1076 output.push_back(std::move(out_chunk));1077 }1078 1079 return true;1080}1081 1082//1083// mtmd_audio_preprocessor_parakeet implementation1084//1085 1086void mtmd_audio_preprocessor_parakeet::worker_thread(1087 int ith,1088 const float * window_func,1089 int window_size,1090 const std::vector<float> & samples,1091 int n_samples,1092 int frame_size,1093 int frame_step,1094 int n_threads,1095 int n_fft_bins,1096 const mtmd_audio_cache & cache,1097 mtmd_audio_mel & mel) {1098 std::vector<float> fft_in(frame_size * 2, 0.0);1099 std::vector<float> fft_out(frame_size * 2 * 2 * 2);1100 1101 int n_fb = n_fft_bins;1102 int i = ith;1103 1104 GGML_ASSERT(n_fb == 1 + (frame_size / 2));1105 1106 const double eps = 5.960464477539063e-08;1107 1108 for (; i < std::min(n_samples / frame_step + 1, (int) mel.n_len); i += n_threads) {1109 const int offset = i * frame_step;1110 const int window_pad_left = (frame_size - window_size) / 2;1111 1112 // Zero-pad left.1113 std::fill(fft_in.begin(), fft_in.begin() + window_pad_left, 0.0f);1114 1115 // Apply windowed samples in the center.1116 const int n_to_process = std::min({window_size, n_samples - offset});1117 for (int j = 0; j < n_to_process; j++) {1118 fft_in[window_pad_left + j] = window_func[j] * samples[offset + window_pad_left + j];1119 }1120 1121 // Zero-pad right.1122 std::fill(fft_in.begin() + window_pad_left + n_to_process, fft_in.begin() + frame_size, 0.0f);1123 1124 // FFT.1125 fft(cache, fft_in.data(), frame_size, fft_out.data());1126 1127 // Calculate modulus^2 of complex numbers.1128 for (int j = 0; j < n_fb; j++) {1129 fft_out[j] = (fft_out[2 * j + 0] * fft_out[2 * j + 0] + fft_out[2 * j + 1] * fft_out[2 * j + 1]);1130 }1131 1132 // mel spectrogram.1133 for (int j = 0; j < mel.n_mel; j++) {1134 double sum = 0.0;1135 int k = 0;1136 for (k = 0; k < n_fb - 3; k += 4) {1137 sum +=1138 fft_out[k + 0] * cache.filters.data[j * n_fb + k + 0] +1139 fft_out[k + 1] * cache.filters.data[j * n_fb + k + 1] +1140 fft_out[k + 2] * cache.filters.data[j * n_fb + k + 2] +1141 fft_out[k + 3] * cache.filters.data[j * n_fb + k + 3];1142 }1143 for (; k < n_fb; k++) {1144 sum += fft_out[k] * cache.filters.data[j * n_fb + k];1145 }1146 mel.data[j * mel.n_len + i] = std::log(sum + eps);1147 }1148 }1149 1150 // Otherwise fft_out are all zero.1151 const double empty_sum = std::log(eps);1152 for (; i < mel.n_len; i += n_threads) {1153 for (int j = 0; j < mel.n_mel; j++) {1154 mel.data[j * mel.n_len + i] = empty_sum;1155 }1156 }1157}1158 1159void mtmd_audio_preprocessor_parakeet::initialize() {1160 cache.fill_sin_cos_table(hparams.audio_n_fft);1161 1162 const size_t n_fft = hparams.audio_n_fft / 2 + 1;1163 GGML_ASSERT(hparams.mel_filters.size() == (size_t)hparams.n_mel_bins * n_fft);1164 cache.filters.n_mel = hparams.n_mel_bins;1165 cache.filters.n_fft = n_fft;1166 cache.filters.data = hparams.mel_filters;1167 1168 GGML_ASSERT(hparams.window.size() == (size_t)hparams.audio_window_len);1169 GGML_ASSERT(hparams.window.size() <= (size_t) hparams.audio_n_fft);1170 cache.hann_window = hparams.window;1171}1172 1173bool mtmd_audio_preprocessor_parakeet::preprocess(const float * samples,1174 size_t n_samples_in,1175 std::vector<mtmd_audio_mel> & output) {1176 if (n_samples_in == 0) {1177 return false;1178 }1179 1180 filter_params params;1181 params.n_mel = hparams.n_mel_bins;1182 params.n_fft_bins = 1 + (hparams.audio_n_fft / 2);1183 params.hann_window_size = hparams.audio_window_len;1184 params.hop_length = hparams.audio_hop_len;1185 params.sample_rate = hparams.audio_sample_rate;1186 1187 GGML_ASSERT(!cache.sin_vals.empty());1188 GGML_ASSERT(!cache.cos_vals.empty());1189 GGML_ASSERT(!cache.filters.data.empty());1190 1191 const float * window_func = cache.hann_window.data();1192 const int window_size = params.hann_window_size;1193 const int frame_size = (params.n_fft_bins - 1) * 2;1194 const int frame_step = params.hop_length;1195 1196 // Apply preemphasis filter (high-pass): x[i] = x[i] - 0.97 * x[i-1]1197 std::vector<float> samples_preprocessed(samples, samples + n_samples_in);1198 {1199 const float preemph = 0.97f;1200 for (int i = n_samples_in - 1; i > 0; i--) {