Team Ai
Apppublic

Xenobd/whisper.cpp

sourceHugging Faceupdated 10mo agoView on Hugging Face
0likes
common-whisper.cpp176 linesDownload Raw Back to examples
1#define _USE_MATH_DEFINES // for M_PI2 3#include "common-whisper.h"4 5#include "common.h"6 7#include "whisper.h"8 9// third-party utilities10// use your favorite implementations11#define STB_VORBIS_HEADER_ONLY12#include "stb_vorbis.c"    /* Enables Vorbis decoding. */13 14#ifdef _WIN3215#ifndef NOMINMAX16    #define NOMINMAX17#endif18#endif19 20#define MA_NO_DEVICE_IO21#define MA_NO_THREADING22#define MA_NO_ENCODING23#define MA_NO_GENERATION24#define MA_NO_RESOURCE_MANAGER25#define MA_NO_NODE_GRAPH26#define MINIAUDIO_IMPLEMENTATION27#include "miniaudio.h"28 29#ifdef _WIN3230#include <fcntl.h>31#include <io.h>32#endif33 34#include <cstring>35#include <fstream>36 37#ifdef WHISPER_FFMPEG38// as implemented in ffmpeg_trancode.cpp only embedded in common lib if whisper built with ffmpeg support39extern bool ffmpeg_decode_audio(const std::string & ifname, std::vector<uint8_t> & wav_data);40#endif41 42bool read_audio_data(const std::string & fname, std::vector<float>& pcmf32, std::vector<std::vector<float>>& pcmf32s, bool stereo) {43    std::vector<uint8_t> audio_data; // used for pipe input from stdin or ffmpeg decoding output44 45    ma_result result;46    ma_decoder_config decoder_config;47    ma_decoder decoder;48 49    decoder_config = ma_decoder_config_init(ma_format_f32, stereo ? 2 : 1, WHISPER_SAMPLE_RATE);50 51    if (fname == "-") {52		#ifdef _WIN3253		_setmode(_fileno(stdin), _O_BINARY);54		#endif55 56		uint8_t buf[1024];57		while (true)58		{59			const size_t n = fread(buf, 1, sizeof(buf), stdin);60			if (n == 0) {61				break;62			}63			audio_data.insert(audio_data.end(), buf, buf + n);64		}65 66		if ((result = ma_decoder_init_memory(audio_data.data(), audio_data.size(), &decoder_config, &decoder)) != MA_SUCCESS) {67 68			fprintf(stderr, "Error: failed to open audio data from stdin (%s)\n", ma_result_description(result));69 70			return false;71		}72 73		fprintf(stderr, "%s: read %zu bytes from stdin\n", __func__, audio_data.size());74    }75    else if (((result = ma_decoder_init_file(fname.c_str(), &decoder_config, &decoder)) != MA_SUCCESS)) {76#if defined(WHISPER_FFMPEG)77		if (ffmpeg_decode_audio(fname, audio_data) != 0) {78			fprintf(stderr, "error: failed to ffmpeg decode '%s'\n", fname.c_str());79 80			return false;81		}82 83		if ((result = ma_decoder_init_memory(audio_data.data(), audio_data.size(), &decoder_config, &decoder)) != MA_SUCCESS) {84			fprintf(stderr, "error: failed to read audio data as wav (%s)\n", ma_result_description(result));85 86			return false;87		}88#else89		if ((result = ma_decoder_init_memory(fname.c_str(), fname.size(), &decoder_config, &decoder)) != MA_SUCCESS) {90			fprintf(stderr, "error: failed to read audio data as wav (%s)\n", ma_result_description(result));91 92			return false;93		}94#endif95    }96 97    ma_uint64 frame_count;98    ma_uint64 frames_read;99 100    if ((result = ma_decoder_get_length_in_pcm_frames(&decoder, &frame_count)) != MA_SUCCESS) {101		fprintf(stderr, "error: failed to retrieve the length of the audio data (%s)\n", ma_result_description(result));102 103		return false;104    }105 106    pcmf32.resize(stereo ? frame_count*2 : frame_count);107 108    if ((result = ma_decoder_read_pcm_frames(&decoder, pcmf32.data(), frame_count, &frames_read)) != MA_SUCCESS) {109		fprintf(stderr, "error: failed to read the frames of the audio data (%s)\n", ma_result_description(result));110 111		return false;112    }113 114    if (stereo) {115        std::vector<float> stereo_data = pcmf32;116        pcmf32.resize(frame_count);117 118        for (uint64_t i = 0; i < frame_count; i++) {119            pcmf32[i] = (stereo_data[2*i] + stereo_data[2*i + 1]);120        }121 122        pcmf32s.resize(2);123        pcmf32s[0].resize(frame_count);124        pcmf32s[1].resize(frame_count);125        for (uint64_t i = 0; i < frame_count; i++) {126            pcmf32s[0][i] = stereo_data[2*i];127            pcmf32s[1][i] = stereo_data[2*i + 1];128        }129    }130 131    ma_decoder_uninit(&decoder);132 133    return true;134}135 136//  500 -> 00:05.000137// 6000 -> 01:00.000138std::string to_timestamp(int64_t t, bool comma) {139    int64_t msec = t * 10;140    int64_t hr = msec / (1000 * 60 * 60);141    msec = msec - hr * (1000 * 60 * 60);142    int64_t min = msec / (1000 * 60);143    msec = msec - min * (1000 * 60);144    int64_t sec = msec / 1000;145    msec = msec - sec * 1000;146 147    char buf[32];148    snprintf(buf, sizeof(buf), "%02d:%02d:%02d%s%03d", (int) hr, (int) min, (int) sec, comma ? "," : ".", (int) msec);149 150    return std::string(buf);151}152 153int timestamp_to_sample(int64_t t, int n_samples, int whisper_sample_rate) {154    return std::max(0, std::min((int) n_samples - 1, (int) ((t*whisper_sample_rate)/100)));155}156 157bool speak_with_file(const std::string & command, const std::string & text, const std::string & path, int voice_id) {158    std::ofstream speak_file(path.c_str());159    if (speak_file.fail()) {160        fprintf(stderr, "%s: failed to open speak_file\n", __func__);161        return false;162    } else {163        speak_file.write(text.c_str(), text.size());164        speak_file.close();165        int ret = system((command + " " + std::to_string(voice_id) + " " + path).c_str());166        if (ret != 0) {167            fprintf(stderr, "%s: failed to speak\n", __func__);168            return false;169        }170    }171    return true;172}173 174#undef STB_VORBIS_HEADER_ONLY175#include "stb_vorbis.c"176