Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#pragma once2 3#include "ggml.h"4#include "clip-model.h"5 6#include <cstdint>7#include <vector>8#include <string>9 10#define MTMD_INTERNAL_HEADER11 12struct mtmd_audio_mel {13 int64_t n_len;14 int64_t n_len_org;15 int64_t n_mel;16 17 std::vector<float> data;18};19 20struct mtmd_audio_mel_filters {21 int64_t n_mel;22 int64_t n_fft;23 24 std::vector<float> data;25};26 27// cache for audio processing, each processor instance owns its own cache28struct mtmd_audio_cache {29 std::vector<float> sin_vals;30 std::vector<float> cos_vals;31 32 std::vector<float> hann_window;33 34 mtmd_audio_mel_filters filters;35 36 void fill_sin_cos_table(uint32_t n);37 38 void fill_hann_window(uint32_t length, bool periodic);39 40 // Build mel filterbank matrix [n_mel × n_fft_bins] at runtime.41 // n_fft_bins must be (N_fft / 2 + 1). Example: if N_fft=512 -> n_fft_bins=257.42 void fill_mel_filterbank_matrix(int64_t n_mel,43 int64_t n_fft,44 int sample_rate, // e.g. 1600045 float fmin = 0.0f, // e.g. 0.046 float fmax = -1.0f, // e.g. sr/2; pass -1 for auto47 bool slaney_area_norm = true,48 float scale = 1.0f,49 bool use_htk = false50 );51};52 53struct mtmd_audio_preprocessor {54 const clip_hparams & hparams;55 56 mtmd_audio_preprocessor(const clip_ctx * ctx): hparams(*clip_get_hparams(ctx)) {}57 58 virtual ~mtmd_audio_preprocessor() = default;59 virtual void initialize() = 0; // NOT thread-safe60 virtual bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) = 0;61};62 63struct mtmd_audio_preprocessor_whisper : mtmd_audio_preprocessor {64 mtmd_audio_preprocessor_whisper(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}65 void initialize() override;66 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;67 68 private:69 mtmd_audio_cache cache;70};71 72struct mtmd_audio_preprocessor_conformer : mtmd_audio_preprocessor {73 mtmd_audio_preprocessor_conformer(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}74 void initialize() override;75 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;76 77 private:78 mtmd_audio_cache cache;79};80 81struct mtmd_audio_preprocessor_granite_speech : mtmd_audio_preprocessor {82 mtmd_audio_preprocessor_granite_speech(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}83 void initialize() override;84 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;85 86 private:87 mtmd_audio_cache cache;88};89 90struct mtmd_audio_preprocessor_gemma4a : mtmd_audio_preprocessor {91 mtmd_audio_preprocessor_gemma4a(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}92 void initialize() override;93 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;94 95 private:96 mtmd_audio_cache cache;97};98 99struct mtmd_audio_preprocessor_gemma4ua : mtmd_audio_preprocessor {100 mtmd_audio_preprocessor_gemma4ua(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}101 void initialize() override;102 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;103};104 105struct mtmd_audio_preprocessor_qwen3a : mtmd_audio_preprocessor {106 mtmd_audio_preprocessor_qwen3a(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}107 void initialize() override;108 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;109 110 private:111 mtmd_audio_cache cache;112};113 114struct mtmd_audio_preprocessor_mimo_audio : mtmd_audio_preprocessor {115 mtmd_audio_preprocessor_mimo_audio(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}116 void initialize() override;117 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;118 119 private:120 mtmd_audio_cache cache;121};122 123struct mtmd_audio_preprocessor_qwen3tts_spk : mtmd_audio_preprocessor {124 mtmd_audio_preprocessor_qwen3tts_spk(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}125 void initialize() override;126 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;127 128 private:129 mtmd_audio_cache cache;130};131 132struct mtmd_audio_preprocessor_parakeet : mtmd_audio_preprocessor {133 mtmd_audio_preprocessor_parakeet(clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) { }134 void initialize() override;135 bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;136 137 private:138 mtmd_audio_cache cache;139 140 static void worker_thread(int ith, const float * window_func, int window_size,141 const std::vector<float> & samples, int n_samples,142 int frame_size, int frame_step, int n_threads,143 int n_fft_bins,144 const mtmd_audio_cache & cache, mtmd_audio_mel & mel);145};146 147//148// streaming ISTFT - converts spectrogram frames back to audio one frame at a time149//150struct mtmd_audio_streaming_istft {151 mtmd_audio_streaming_istft(int n_fft, int hop_length);152 153 // reset streaming state154 void reset();155 156 // process a single STFT frame (streaming)157 // frame_spectrum: [n_fft_bins x 2] interleaved real/imag158 // returns: up to hop_length samples159 std::vector<float> process_frame(const float * frame_spectrum);160 161 // flush remaining samples at end of stream162 std::vector<float> flush();163 164 private:165 int n_fft;166 int hop_length;167 int n_fft_bins;168 169 // Own cache for output processing170 mtmd_audio_cache cache;171 172 // Streaming state173 std::vector<float> overlap_buffer;174 std::vector<float> window_sum_buffer;175 int padding_to_remove;176 177 // Working buffers for IFFT178 std::vector<float> ifft_in;179 std::vector<float> ifft_out;180};181 