Xenobd/whisper.cpp
0
1// Various helper functions and utilities2 3#pragma once4 5#include <string>6#include <map>7#include <vector>8#include <random>9#include <thread>10#include <ctime>11#include <fstream>12#include <sstream>13 14//15// GPT CLI argument parsing16//17 18struct gpt_params {19 int32_t seed = -1; // RNG seed20 int32_t n_threads = std::min(4, (int32_t) std::thread::hardware_concurrency());21 int32_t n_predict = 200; // new tokens to predict22 int32_t n_parallel = 1; // number of parallel streams23 int32_t n_batch = 32; // batch size for prompt processing24 int32_t n_ctx = 2048; // context size (this is the KV cache max size)25 int32_t n_gpu_layers = 0; // number of layers to offlload to the GPU26 27 bool ignore_eos = false; // ignore EOS token when generating text28 29 // sampling parameters30 int32_t top_k = 40;31 float top_p = 0.9f;32 float temp = 0.9f;33 int32_t repeat_last_n = 64;34 float repeat_penalty = 1.00f;35 36 std::string model = "models/gpt-2-117M/ggml-model.bin"; // model path37 std::string prompt = "";38 std::string token_test = "";39 40 bool interactive = false;41 int32_t interactive_port = -1;42};43 44bool gpt_params_parse(int argc, char ** argv, gpt_params & params);45 46void gpt_print_usage(int argc, char ** argv, const gpt_params & params);47 48std::string gpt_random_prompt(std::mt19937 & rng);49 50//51// Vocab utils52//53 54std::string trim(const std::string & s);55 56std::string replace(57 const std::string & s,58 const std::string & from,59 const std::string & to);60 61struct gpt_vocab {62 using id = int32_t;63 using token = std::string;64 65 std::map<token, id> token_to_id;66 std::map<id, token> id_to_token;67 std::vector<std::string> special_tokens;68 69 void add_special_token(const std::string & token);70};71 72// poor-man's JSON parsing73std::map<std::string, int32_t> json_parse(const std::string & fname);74 75std::string convert_to_utf8(const std::wstring & input);76 77std::wstring convert_to_wstring(const std::string & input);78 79void gpt_split_words(std::string str, std::vector<std::string>& words);80 81// split text into tokens82//83// ref: https://github.com/openai/gpt-2/blob/a74da5d99abaaba920de8131d64da2862a8f213b/src/encoder.py#L5384//85// Regex (Python):86// r"""'s|'t|'re|'ve|'m|'ll|'d| ?\p{L}+| ?\p{N}+| ?[^\s\p{L}\p{N}]+|\s+(?!\S)|\s+"""87//88// Regex (C++):89// R"('s|'t|'re|'ve|'m|'ll|'d| ?[[:alpha:]]+| ?[[:digit:]]+| ?[^\s[:alpha:][:digit:]]+|\s+(?!\S)|\s+)"90//91std::vector<gpt_vocab::id> gpt_tokenize(const gpt_vocab & vocab, const std::string & text);92 93// test outputs of gpt_tokenize94//95// - compare with tokens generated by the huggingface tokenizer96// - test cases are chosen based on the model's main language (under 'prompt' directory)97// - if all sentences are tokenized identically, print 'All tests passed.'98// - otherwise, print sentence, huggingface tokens, ggml tokens99//100void test_gpt_tokenizer(gpt_vocab & vocab, const std::string & fpath_test);101 102// load the tokens from encoder.json103bool gpt_vocab_init(const std::string & fname, gpt_vocab & vocab);104 105// sample next token given probabilities for each embedding106//107// - consider only the top K tokens108// - from them, consider only the top tokens with cumulative probability > P109//110// TODO: not sure if this implementation is correct111// TODO: temperature is not implemented112//113gpt_vocab::id gpt_sample_top_k_top_p(114 const gpt_vocab & vocab,115 const float * logits,116 int top_k,117 double top_p,118 double temp,119 std::mt19937 & rng);120 121gpt_vocab::id gpt_sample_top_k_top_p_repeat(122 const gpt_vocab & vocab,123 const float * logits,124 const int32_t * last_n_tokens_data,125 size_t last_n_tokens_data_size,126 int top_k,127 double top_p,128 double temp,129 int repeat_last_n,130 float repeat_penalty,131 std::mt19937 & rng);132 133//134// Audio utils135//136 137// Write PCM data into WAV audio file138class wav_writer {139private:140 std::ofstream file;141 uint32_t dataSize = 0;142 std::string wav_filename;143 144 bool write_header(const uint32_t sample_rate,145 const uint16_t bits_per_sample,146 const uint16_t channels) {147 148 file.write("RIFF", 4);149 file.write("\0\0\0\0", 4); // Placeholder for file size150 file.write("WAVE", 4);151 file.write("fmt ", 4);152 153 const uint32_t sub_chunk_size = 16;154 const uint16_t audio_format = 1; // PCM format155 const uint32_t byte_rate = sample_rate * channels * bits_per_sample / 8;156 const uint16_t block_align = channels * bits_per_sample / 8;157 158 file.write(reinterpret_cast<const char *>(&sub_chunk_size), 4);159 file.write(reinterpret_cast<const char *>(&audio_format), 2);160 file.write(reinterpret_cast<const char *>(&channels), 2);161 file.write(reinterpret_cast<const char *>(&sample_rate), 4);162 file.write(reinterpret_cast<const char *>(&byte_rate), 4);163 file.write(reinterpret_cast<const char *>(&block_align), 2);164 file.write(reinterpret_cast<const char *>(&bits_per_sample), 2);165 file.write("data", 4);166 file.write("\0\0\0\0", 4); // Placeholder for data size167 168 return true;169 }170 171 // It is assumed that PCM data is normalized to a range from -1 to 1172 bool write_audio(const float * data, size_t length) {173 for (size_t i = 0; i < length; ++i) {174 const int16_t intSample = int16_t(data[i] * 32767);175 file.write(reinterpret_cast<const char *>(&intSample), sizeof(int16_t));176 dataSize += sizeof(int16_t);177 }178 if (file.is_open()) {179 file.seekp(4, std::ios::beg);180 uint32_t fileSize = 36 + dataSize;181 file.write(reinterpret_cast<char *>(&fileSize), 4);182 file.seekp(40, std::ios::beg);183 file.write(reinterpret_cast<char *>(&dataSize), 4);184 file.seekp(0, std::ios::end);185 }186 return true;187 }188 189 bool open_wav(const std::string & filename) {190 if (filename != wav_filename) {191 if (file.is_open()) {192 file.close();193 }194 }195 if (!file.is_open()) {196 file.open(filename, std::ios::binary);197 wav_filename = filename;198 dataSize = 0;199 }200 return file.is_open();201 }202 203public:204 bool open(const std::string & filename,205 const uint32_t sample_rate,206 const uint16_t bits_per_sample,207 const uint16_t channels) {208 209 if (open_wav(filename)) {210 write_header(sample_rate, bits_per_sample, channels);211 } else {212 return false;213 }214 215 return true;216 }217 218 bool close() {219 file.close();220 return true;221 }222 223 bool write(const float * data, size_t length) {224 return write_audio(data, length);225 }226 227 ~wav_writer() {228 if (file.is_open()) {229 file.close();230 }231 }232};233 234 235// Apply a high-pass frequency filter to PCM audio236// Suppresses frequencies below cutoff Hz237void high_pass_filter(238 std::vector<float> & data,239 float cutoff,240 float sample_rate);241 242// Basic voice activity detection (VAD) using audio energy adaptive threshold243bool vad_simple(244 std::vector<float> & pcmf32,245 int sample_rate,246 int last_ms,247 float vad_thold,248 float freq_thold,249 bool verbose);250 251// compute similarity between two strings using Levenshtein distance252float similarity(const std::string & s0, const std::string & s1);253 254//255// Terminal utils256//257 258#define SQR(X) ((X) * (X))259#define UNCUBE(x) x < 48 ? 0 : x < 115 ? 1 : (x - 35) / 40260 261/**262 * Quantizes 24-bit RGB to xterm256 code range [16,256).263 */264static int rgb2xterm256(int r, int g, int b) {265 unsigned char cube[] = {0, 0137, 0207, 0257, 0327, 0377};266 int av, ir, ig, ib, il, qr, qg, qb, ql;267 av = r * .299 + g * .587 + b * .114 + .5;268 ql = (il = av > 238 ? 23 : (av - 3) / 10) * 10 + 8;269 qr = cube[(ir = UNCUBE(r))];270 qg = cube[(ig = UNCUBE(g))];271 qb = cube[(ib = UNCUBE(b))];272 if (SQR(qr - r) + SQR(qg - g) + SQR(qb - b) <=273 SQR(ql - r) + SQR(ql - g) + SQR(ql - b))274 return ir * 36 + ig * 6 + ib + 020;275 return il + 0350;276}277 278static std::string set_xterm256_foreground(int r, int g, int b) {279 int x = rgb2xterm256(r, g, b);280 std::ostringstream oss;281 oss << "\033[38;5;" << x << "m";282 return oss.str();283}284 285// Lowest is red, middle is yellow, highest is green. Color scheme from286// Paul Tol; it is colorblind friendly https://sronpersonalpages.nl/~pault287const std::vector<std::string> k_colors = {288 set_xterm256_foreground(220, 5, 12),289 set_xterm256_foreground(232, 96, 28),290 set_xterm256_foreground(241, 147, 45),291 set_xterm256_foreground(246, 193, 65),292 set_xterm256_foreground(247, 240, 86),293 set_xterm256_foreground(144, 201, 135),294 set_xterm256_foreground( 78, 178, 101),295};296 297// ANSI formatting codes298static std::string set_inverse() {299 return "\033[7m";300}301 302static std::string set_underline() {303 return "\033[4m";304}305 306static std::string set_dim() {307 return "\033[2m";308}309 310// Style scheme for different confidence levels311const std::vector<std::string> k_styles = {312 set_inverse(), // Low confidence - inverse (highlighted)313 set_underline(), // Medium confidence - underlined314 set_dim(), // High confidence - dim315};316 317//318// Other utils319//320 321// check if file exists using ifstream322bool is_file_exist(const char * filename);323 