Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
mtmd-audio.h181 linesDownload Raw Back to mtmd
1#pragma once2 3#include "ggml.h"4#include "clip-model.h"5 6#include <cstdint>7#include <vector>8#include <string>9 10#define MTMD_INTERNAL_HEADER11 12struct mtmd_audio_mel {13    int64_t n_len;14    int64_t n_len_org;15    int64_t n_mel;16 17    std::vector<float> data;18};19 20struct mtmd_audio_mel_filters {21    int64_t n_mel;22    int64_t n_fft;23 24    std::vector<float> data;25};26 27// cache for audio processing, each processor instance owns its own cache28struct mtmd_audio_cache {29    std::vector<float> sin_vals;30    std::vector<float> cos_vals;31 32    std::vector<float> hann_window;33 34    mtmd_audio_mel_filters filters;35 36    void fill_sin_cos_table(uint32_t n);37 38    void fill_hann_window(uint32_t length, bool periodic);39 40    // Build mel filterbank matrix [n_mel × n_fft_bins] at runtime.41    // n_fft_bins must be (N_fft / 2 + 1). Example: if N_fft=512 -> n_fft_bins=257.42    void fill_mel_filterbank_matrix(int64_t n_mel,43                                    int64_t n_fft,44                                    int   sample_rate,               // e.g. 1600045                                    float fmin             = 0.0f,   // e.g. 0.046                                    float fmax             = -1.0f,  // e.g. sr/2; pass -1 for auto47                                    bool  slaney_area_norm = true,48                                    float scale            = 1.0f,49                                    bool  use_htk          = false50    );51};52 53struct mtmd_audio_preprocessor {54    const clip_hparams & hparams;55 56    mtmd_audio_preprocessor(const clip_ctx * ctx): hparams(*clip_get_hparams(ctx)) {}57 58    virtual ~mtmd_audio_preprocessor() = default;59    virtual void initialize() = 0; // NOT thread-safe60    virtual bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) = 0;61};62 63struct mtmd_audio_preprocessor_whisper : mtmd_audio_preprocessor {64    mtmd_audio_preprocessor_whisper(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}65    void initialize() override;66    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;67 68  private:69    mtmd_audio_cache cache;70};71 72struct mtmd_audio_preprocessor_conformer : mtmd_audio_preprocessor {73    mtmd_audio_preprocessor_conformer(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}74    void initialize() override;75    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;76 77  private:78    mtmd_audio_cache cache;79};80 81struct mtmd_audio_preprocessor_granite_speech : mtmd_audio_preprocessor {82    mtmd_audio_preprocessor_granite_speech(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}83    void initialize() override;84    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;85 86  private:87    mtmd_audio_cache cache;88};89 90struct mtmd_audio_preprocessor_gemma4a : mtmd_audio_preprocessor {91    mtmd_audio_preprocessor_gemma4a(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}92    void initialize() override;93    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;94 95  private:96    mtmd_audio_cache cache;97};98 99struct mtmd_audio_preprocessor_gemma4ua : mtmd_audio_preprocessor {100    mtmd_audio_preprocessor_gemma4ua(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}101    void initialize() override;102    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;103};104 105struct mtmd_audio_preprocessor_qwen3a : mtmd_audio_preprocessor {106    mtmd_audio_preprocessor_qwen3a(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}107    void initialize() override;108    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;109 110  private:111    mtmd_audio_cache cache;112};113 114struct mtmd_audio_preprocessor_mimo_audio : mtmd_audio_preprocessor {115    mtmd_audio_preprocessor_mimo_audio(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}116    void initialize() override;117    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;118 119  private:120    mtmd_audio_cache cache;121};122 123struct mtmd_audio_preprocessor_qwen3tts_spk : mtmd_audio_preprocessor {124    mtmd_audio_preprocessor_qwen3tts_spk(const clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) {}125    void initialize() override;126    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;127 128  private:129    mtmd_audio_cache cache;130};131 132struct mtmd_audio_preprocessor_parakeet : mtmd_audio_preprocessor {133    mtmd_audio_preprocessor_parakeet(clip_ctx * ctx) : mtmd_audio_preprocessor(ctx) { }134    void initialize() override;135    bool preprocess(const float * samples, size_t n_samples, std::vector<mtmd_audio_mel> & output) override;136 137  private:138    mtmd_audio_cache cache;139 140    static void worker_thread(int ith, const float * window_func, int window_size,141                              const std::vector<float> & samples, int n_samples,142                              int frame_size, int frame_step, int n_threads,143                              int n_fft_bins,144                              const mtmd_audio_cache & cache, mtmd_audio_mel & mel);145};146 147//148// streaming ISTFT - converts spectrogram frames back to audio one frame at a time149//150struct mtmd_audio_streaming_istft {151    mtmd_audio_streaming_istft(int n_fft, int hop_length);152 153    // reset streaming state154    void reset();155 156    // process a single STFT frame (streaming)157    // frame_spectrum: [n_fft_bins x 2] interleaved real/imag158    // returns: up to hop_length samples159    std::vector<float> process_frame(const float * frame_spectrum);160 161    // flush remaining samples at end of stream162    std::vector<float> flush();163 164  private:165    int n_fft;166    int hop_length;167    int n_fft_bins;168 169    // Own cache for output processing170    mtmd_audio_cache cache;171 172    // Streaming state173    std::vector<float> overlap_buffer;174    std::vector<float> window_sum_buffer;175    int                padding_to_remove;176 177    // Working buffers for IFFT178    std::vector<float> ifft_in;179    std::vector<float> ifft_out;180};181 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai