natasa365/whisper.cpp
0
1#include "whisper.h"2#include "common-whisper.h"3 4#include <cstdio>5#include <cfloat>6#include <string>7#include <cstring>8 9#ifdef NDEBUG10#undef NDEBUG11#endif12 13#include <cassert>14 15int main() {16 std::string whisper_model_path = "../../models/ggml-base.en.bin";17 std::string vad_model_path = "../../models/for-tests-silero-v5.1.2-ggml.bin";18 std::string sample_path = "../../samples/jfk.wav";19 20 // Load the sample audio file21 std::vector<float> pcmf32;22 std::vector<std::vector<float>> pcmf32s;23 assert(read_audio_data(sample_path.c_str(), pcmf32, pcmf32s, false));24 25 struct whisper_context_params cparams = whisper_context_default_params();26 struct whisper_context * wctx = whisper_init_from_file_with_params(27 whisper_model_path.c_str(),28 cparams);29 30 struct whisper_full_params wparams = whisper_full_default_params(WHISPER_SAMPLING_BEAM_SEARCH);31 wparams.vad = true;32 wparams.vad_model_path = vad_model_path.c_str();33 34 wparams.vad_params.threshold = 0.5f;35 wparams.vad_params.min_speech_duration_ms = 250;36 wparams.vad_params.min_silence_duration_ms = 100;37 wparams.vad_params.max_speech_duration_s = FLT_MAX;38 wparams.vad_params.speech_pad_ms = 30;39 40 assert(whisper_full_parallel(wctx, wparams, pcmf32.data(), pcmf32.size(), 1) == 0);41 42 const int n_segments = whisper_full_n_segments(wctx);43 assert(n_segments == 1);44 45 assert(strcmp(" And so my fellow Americans, ask not what your country can do for you,"46 " ask what you can do for your country.",47 whisper_full_get_segment_text(wctx, 0)) == 0);48 assert(whisper_full_get_segment_t0(wctx, 0) == 29);49 assert(whisper_full_get_segment_t1(wctx, 0) == 1049);50 51 whisper_free(wctx);52 53 return 0;54}55 