Team Ai
Apppublic

natasa365/whisper.cpp

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
test-vad-full.cpp55 linesDownload Raw Back to tests
1#include "whisper.h"2#include "common-whisper.h"3 4#include <cstdio>5#include <cfloat>6#include <string>7#include <cstring>8 9#ifdef NDEBUG10#undef NDEBUG11#endif12 13#include <cassert>14 15int main() {16    std::string whisper_model_path = "../../models/ggml-base.en.bin";17    std::string vad_model_path     = "../../models/for-tests-silero-v5.1.2-ggml.bin";18    std::string sample_path        = "../../samples/jfk.wav";19 20    // Load the sample audio file21    std::vector<float> pcmf32;22    std::vector<std::vector<float>> pcmf32s;23    assert(read_audio_data(sample_path.c_str(), pcmf32, pcmf32s, false));24 25    struct whisper_context_params cparams = whisper_context_default_params();26    struct whisper_context * wctx = whisper_init_from_file_with_params(27            whisper_model_path.c_str(),28            cparams);29 30    struct whisper_full_params wparams = whisper_full_default_params(WHISPER_SAMPLING_BEAM_SEARCH);31    wparams.vad            = true;32    wparams.vad_model_path = vad_model_path.c_str();33 34    wparams.vad_params.threshold               = 0.5f;35    wparams.vad_params.min_speech_duration_ms  = 250;36    wparams.vad_params.min_silence_duration_ms = 100;37    wparams.vad_params.max_speech_duration_s   = FLT_MAX;38    wparams.vad_params.speech_pad_ms           = 30;39 40    assert(whisper_full_parallel(wctx, wparams, pcmf32.data(), pcmf32.size(), 1) == 0);41 42    const int n_segments = whisper_full_n_segments(wctx);43    assert(n_segments == 1);44 45    assert(strcmp(" And so my fellow Americans, ask not what your country can do for you,"46                  " ask what you can do for your country.",47           whisper_full_get_segment_text(wctx, 0)) == 0);48    assert(whisper_full_get_segment_t0(wctx, 0) == 29);49    assert(whisper_full_get_segment_t1(wctx, 0) == 1049);50 51    whisper_free(wctx);52 53    return 0;54}55