Team Ai
Apppublic

Xenobd/whisper.cpp

sourceHugging Faceupdated 10mo agoView on Hugging Face
0likes
test-vad.cpp84 linesDownload Raw Back to tests
1#include "whisper.h"2#include "common-whisper.h"3 4#include <cstdio>5#include <string>6 7#ifdef NDEBUG8#undef NDEBUG9#endif10#include <cassert>11 12void assert_default_params(const struct whisper_vad_params & params) {13    assert(params.threshold == 0.5);14    assert(params.min_speech_duration_ms == 250);15    assert(params.min_silence_duration_ms == 100);16    assert(params.samples_overlap == 0.1f);17}18 19void assert_default_context_params(const struct whisper_vad_context_params & params) {20    assert(params.n_threads == 4);21    assert(params.use_gpu == false);22    assert(params.gpu_device == 0);23}24 25void test_detect_speech(26        struct whisper_vad_context * vctx,27        struct whisper_vad_params params,28        const float * pcmf32,29        int n_samples) {30    assert(whisper_vad_detect_speech(vctx, pcmf32, n_samples));31    assert(whisper_vad_n_probs(vctx) == 344);32    assert(whisper_vad_probs(vctx) != nullptr);33}34 35struct whisper_vad_segments * test_detect_timestamps(36        struct whisper_vad_context * vctx,37        struct whisper_vad_params params) {38    struct whisper_vad_segments * timestamps = whisper_vad_segments_from_probs(vctx, params);39    assert(whisper_vad_segments_n_segments(timestamps) == 5);40 41    for (int i = 0; i < whisper_vad_segments_n_segments(timestamps); ++i) {42        printf("VAD segment %d: start = %.2f, end = %.2f\n", i,43               whisper_vad_segments_get_segment_t0(timestamps, i),44               whisper_vad_segments_get_segment_t1(timestamps, i));45    }46 47    return timestamps;48}49 50int main() {51    std::string vad_model_path = "../../models/for-tests-silero-v5.1.2-ggml.bin";52    std::string sample_path    = "../../samples/jfk.wav";53 54    // Load the sample audio file55    std::vector<float> pcmf32;56    std::vector<std::vector<float>> pcmf32s;57    assert(read_audio_data(sample_path.c_str(), pcmf32, pcmf32s, false));58    assert(pcmf32.size() > 0);59    assert(pcmf32s.size() == 0); // no stereo vector60 61    // Load the VAD model62    struct whisper_vad_context_params ctx_params = whisper_vad_default_context_params();63    assert_default_context_params(ctx_params);64 65    struct whisper_vad_context * vctx = whisper_vad_init_from_file_with_params(66            vad_model_path.c_str(),67            ctx_params);68    assert(vctx != nullptr);69 70    struct whisper_vad_params params = whisper_vad_default_params();71    assert_default_params(params);72 73    // Test speech probabilites74    test_detect_speech(vctx, params, pcmf32.data(), pcmf32.size());75 76    // Test speech timestamps (uses speech probabilities from above)77    struct whisper_vad_segments * timestamps = test_detect_timestamps(vctx, params);78 79    whisper_vad_free_segments(timestamps);80    whisper_vad_free(vctx);81 82    return 0;83}84