Xenobd/whisper.cpp
0
1#define _USE_MATH_DEFINES // for M_PI2 3#include "common-whisper.h"4 5#include "common.h"6 7#include "whisper.h"8 9// third-party utilities10// use your favorite implementations11#define STB_VORBIS_HEADER_ONLY12#include "stb_vorbis.c" /* Enables Vorbis decoding. */13 14#ifdef _WIN3215#ifndef NOMINMAX16 #define NOMINMAX17#endif18#endif19 20#define MA_NO_DEVICE_IO21#define MA_NO_THREADING22#define MA_NO_ENCODING23#define MA_NO_GENERATION24#define MA_NO_RESOURCE_MANAGER25#define MA_NO_NODE_GRAPH26#define MINIAUDIO_IMPLEMENTATION27#include "miniaudio.h"28 29#ifdef _WIN3230#include <fcntl.h>31#include <io.h>32#endif33 34#include <cstring>35#include <fstream>36 37#ifdef WHISPER_FFMPEG38// as implemented in ffmpeg_trancode.cpp only embedded in common lib if whisper built with ffmpeg support39extern bool ffmpeg_decode_audio(const std::string & ifname, std::vector<uint8_t> & wav_data);40#endif41 42bool read_audio_data(const std::string & fname, std::vector<float>& pcmf32, std::vector<std::vector<float>>& pcmf32s, bool stereo) {43 std::vector<uint8_t> audio_data; // used for pipe input from stdin or ffmpeg decoding output44 45 ma_result result;46 ma_decoder_config decoder_config;47 ma_decoder decoder;48 49 decoder_config = ma_decoder_config_init(ma_format_f32, stereo ? 2 : 1, WHISPER_SAMPLE_RATE);50 51 if (fname == "-") {52 #ifdef _WIN3253 _setmode(_fileno(stdin), _O_BINARY);54 #endif55 56 uint8_t buf[1024];57 while (true)58 {59 const size_t n = fread(buf, 1, sizeof(buf), stdin);60 if (n == 0) {61 break;62 }63 audio_data.insert(audio_data.end(), buf, buf + n);64 }65 66 if ((result = ma_decoder_init_memory(audio_data.data(), audio_data.size(), &decoder_config, &decoder)) != MA_SUCCESS) {67 68 fprintf(stderr, "Error: failed to open audio data from stdin (%s)\n", ma_result_description(result));69 70 return false;71 }72 73 fprintf(stderr, "%s: read %zu bytes from stdin\n", __func__, audio_data.size());74 }75 else if (((result = ma_decoder_init_file(fname.c_str(), &decoder_config, &decoder)) != MA_SUCCESS)) {76#if defined(WHISPER_FFMPEG)77 if (ffmpeg_decode_audio(fname, audio_data) != 0) {78 fprintf(stderr, "error: failed to ffmpeg decode '%s'\n", fname.c_str());79 80 return false;81 }82 83 if ((result = ma_decoder_init_memory(audio_data.data(), audio_data.size(), &decoder_config, &decoder)) != MA_SUCCESS) {84 fprintf(stderr, "error: failed to read audio data as wav (%s)\n", ma_result_description(result));85 86 return false;87 }88#else89 if ((result = ma_decoder_init_memory(fname.c_str(), fname.size(), &decoder_config, &decoder)) != MA_SUCCESS) {90 fprintf(stderr, "error: failed to read audio data as wav (%s)\n", ma_result_description(result));91 92 return false;93 }94#endif95 }96 97 ma_uint64 frame_count;98 ma_uint64 frames_read;99 100 if ((result = ma_decoder_get_length_in_pcm_frames(&decoder, &frame_count)) != MA_SUCCESS) {101 fprintf(stderr, "error: failed to retrieve the length of the audio data (%s)\n", ma_result_description(result));102 103 return false;104 }105 106 pcmf32.resize(stereo ? frame_count*2 : frame_count);107 108 if ((result = ma_decoder_read_pcm_frames(&decoder, pcmf32.data(), frame_count, &frames_read)) != MA_SUCCESS) {109 fprintf(stderr, "error: failed to read the frames of the audio data (%s)\n", ma_result_description(result));110 111 return false;112 }113 114 if (stereo) {115 std::vector<float> stereo_data = pcmf32;116 pcmf32.resize(frame_count);117 118 for (uint64_t i = 0; i < frame_count; i++) {119 pcmf32[i] = (stereo_data[2*i] + stereo_data[2*i + 1]);120 }121 122 pcmf32s.resize(2);123 pcmf32s[0].resize(frame_count);124 pcmf32s[1].resize(frame_count);125 for (uint64_t i = 0; i < frame_count; i++) {126 pcmf32s[0][i] = stereo_data[2*i];127 pcmf32s[1][i] = stereo_data[2*i + 1];128 }129 }130 131 ma_decoder_uninit(&decoder);132 133 return true;134}135 136// 500 -> 00:05.000137// 6000 -> 01:00.000138std::string to_timestamp(int64_t t, bool comma) {139 int64_t msec = t * 10;140 int64_t hr = msec / (1000 * 60 * 60);141 msec = msec - hr * (1000 * 60 * 60);142 int64_t min = msec / (1000 * 60);143 msec = msec - min * (1000 * 60);144 int64_t sec = msec / 1000;145 msec = msec - sec * 1000;146 147 char buf[32];148 snprintf(buf, sizeof(buf), "%02d:%02d:%02d%s%03d", (int) hr, (int) min, (int) sec, comma ? "," : ".", (int) msec);149 150 return std::string(buf);151}152 153int timestamp_to_sample(int64_t t, int n_samples, int whisper_sample_rate) {154 return std::max(0, std::min((int) n_samples - 1, (int) ((t*whisper_sample_rate)/100)));155}156 157bool speak_with_file(const std::string & command, const std::string & text, const std::string & path, int voice_id) {158 std::ofstream speak_file(path.c_str());159 if (speak_file.fail()) {160 fprintf(stderr, "%s: failed to open speak_file\n", __func__);161 return false;162 } else {163 speak_file.write(text.c_str(), text.size());164 speak_file.close();165 int ret = system((command + " " + std::to_string(voice_id) + " " + path).c_str());166 if (ret != 0) {167 fprintf(stderr, "%s: failed to speak\n", __func__);168 return false;169 }170 }171 return true;172}173 174#undef STB_VORBIS_HEADER_ONLY175#include "stb_vorbis.c"176 