Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#include "arg.h"2#include "common.h"3#include "sampling.h"4#include "log.h"5#include "llama.h"6#include "mtmd.h"7#include "mtmd-helper.h"8 9#include <cstdio>10#include <cstring>11#include <string>12 13/**14 * Please note that this is NOT a production-ready binary.15 * It is a playground for trying TTS support in llama.cpp.16 * For contributors: please keep this code simple and easy to understand. Do not add unnecessary complexity. The goal is to have a simple CLI for testing TTS support.17 */18 19struct tts_timings {20 int64_t t_start_us = ggml_time_us();21 int64_t t_last_us = t_start_us;22 23 void report(int n_frames) {24 const int64_t t_now_us = ggml_time_us();25 if (t_now_us - t_last_us < 2000000) {26 return;27 }28 t_last_us = t_now_us;29 const double t_elapsed_s = (t_now_us - t_start_us) / 1e6;30 const double fps = t_elapsed_s > 0 ? n_frames / t_elapsed_s : 0.0;31 LOG_INF("frames generated: %d, speed: %.2f frames/s\n", n_frames, fps);32 }33};34 35static void print_usage(int, char ** argv) {36 LOG("\nexample usage:\n");37 LOG("\n %s -m backbone.gguf -mm mmproj.gguf -p \"text to speak\" -o output.wav", argv[0]);38 LOG("\n %s -hf user/model -p \"text to speak\" -o output.wav\n", argv[0]);39 LOG("\nnote: --tts-lang and --tts-speaker-file may not be supported in all models");40 LOG("\n use -n to limit the output length");41 LOG("\n see tts/README.md for per-model usage notes");42 LOG("\n\n");43}44 45int main(int argc, char ** argv) {46 common_params params;47 48 common_init();49 50 if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_TTS, print_usage)) {51 return 1;52 }53 54 mtmd_helper_log_set(common_log_default_callback, nullptr);55 56 if (params.prompt.empty()) {57 LOG_ERR("no prompt provided, use -p \"text\"\n");58 return 1;59 }60 if (params.mmproj.path.empty()) {61 LOG_ERR("no mmproj provided, use --mmproj\n");62 return 1;63 }64 65 // important: keep this file as generic as possible66 // model-specific logic should be in mtmd-helper-gen or mtmd API67 68 // always enable embd, so that we can pass hidden states to the audio generation helper69 params.embedding = true;70 71 llama_backend_init();72 llama_numa_init(params.numa);73 74 //75 // load backbone model and mmproj76 //77 78 auto llama_init = common_init_from_params(params);79 llama_model * model = llama_init->model();80 llama_context * lctx = llama_init->context();81 common_sampler * smpl = llama_init->sampler(0);82 if (!model || !lctx) {83 LOG_ERR("failed to init model/context\n");84 return 1;85 }86 87 mtmd_context_params mtmd_params = mtmd_context_params_default();88 mtmd_params.use_gpu = params.mmproj_use_gpu;89 mtmd::context_ptr mctx(mtmd_init_from_file(params.mmproj.path.c_str(), model, mtmd_params));90 if (!mctx) {91 LOG_ERR("failed to load mmproj %s\n", params.mmproj.path.c_str());92 return 1;93 }94 if (mtmd_gen_audio_get_info(mctx.get()).type == MTMD_GEN_AUDIO_TYPE_NONE) {95 LOG_ERR("mmproj does not support audio generation\n");96 return 1;97 }98 99 //100 // stage 0: process speaker reference file, if any101 //102 103 mtmd::bitmap_ptr speaker_bitmap;104 if (!params.tts_speaker_file.empty()) {105 auto wrapper = mtmd_helper_bitmap_init_from_file(mctx.get(), params.tts_speaker_file.c_str(), false);106 if (!wrapper.bitmap) {107 LOG_ERR("failed to load speaker file %s\n", params.tts_speaker_file.c_str());108 return 1;109 }110 speaker_bitmap.reset(wrapper.bitmap);111 }112 113 mtmd_helper::gen_audio gen(lctx, mctx.get());114 mtmd_helper_gen_audio_inp inp{};115 inp.seq_id = 0;116 inp.prompt = params.prompt.c_str();117 inp.prompt_len = params.prompt.size();118 inp.speaker_ref = speaker_bitmap.get();119 inp.lang = params.tts_lang.c_str();120 inp.top_k = params.sampling.top_k;121 inp.top_p = params.sampling.top_p;122 inp.out_type = MTMD_HELPER_GEN_AUDIO_OUTTYPE_WAV;123 124 //125 // stage 1: process prompt via backbone model, generate semantic representation126 //127 128 if (gen.set_input(&inp) != 0) {129 LOG_ERR("set_input failed\n");130 return 1;131 }132 133 const int64_t t_prompt_start_us = ggml_time_us();134 135 for (;;) {136 int32_t ret = gen.step_prompt(params.n_batch);137 if (ret < 0) {138 LOG_ERR("prompt processing failed\n");139 return 1;140 }141 if (ret == 0) {142 break;143 }144 }145 146 const llama_vocab * vocab = llama_model_get_vocab(model);147 148 auto sample_semantic_code = [&]() -> llama_token {149 llama_token t = common_sampler_sample(smpl, lctx, -1);150 common_sampler_accept(smpl, t, true);151 return t;152 };153 154 const int max_new = params.n_predict > 0 ? params.n_predict : 512;155 int n_frames = 0;156 llama_token sampled = sample_semantic_code();157 const float * h_state = llama_get_embeddings_ith(lctx, -1);158 159 tts_timings timings;160 const int64_t t_gen_start_us = ggml_time_us();161 162 for (; n_frames < max_new && !llama_vocab_is_eog(vocab, sampled); n_frames++) {163 const float * h_next = nullptr;164 165 // stage 2+3: semantic --> acoustic details --> audio waveform166 // step_gen() runs both stages and returns new h_state for next step167 if (gen.step_gen(sampled, h_state, &h_next) != 0) {168 LOG_ERR("step_gen failed at frame %d\n", n_frames);169 return 1;170 }171 172 h_state = h_next;173 sampled = sample_semantic_code();174 timings.report(n_frames + 1);175 }176 const double t_gen_s = (ggml_time_us() - t_gen_start_us) / 1e6;177 178 int32_t sample_rate = 0;179 const char * data = nullptr;180 size_t data_len = 0;181 int64_t n_samples = 0;182 const int64_t t_wav_start_us = ggml_time_us();183 if (gen.get_output(&sample_rate, &data, &data_len, &n_samples) != 0) {184 LOG_ERR("get_output failed\n");185 return 1;186 }187 const double t_wav_s = (ggml_time_us() - t_wav_start_us) / 1e6;188 189 LOG_INF("generated %d frames, %zu bytes of WAV audio (%d Hz)\n", n_frames, data_len, sample_rate);190 191 const double t_prompt_s = (t_gen_start_us - t_prompt_start_us) / 1e6;192 const double t_total_s = t_prompt_s + t_gen_s + t_wav_s;193 const double audio_s = sample_rate > 0 ? (double) n_samples / sample_rate : 0.0;194 LOG_INF("timings: prompt eval %.2fs + generation %.2fs + vocoder %.2fs = total %.2fs\n",195 t_prompt_s, t_gen_s, t_wav_s, t_total_s);196 LOG_INF(" output audio = %.2fs (audio time = %.2fx process time)\n", audio_s, t_total_s > 0 ? audio_s / t_total_s : 0.0);197 FILE * f = fopen(params.out_file.c_str(), "wb");198 if (!f) {199 LOG_ERR("failed to open %s\n", params.out_file.c_str());200 return 1;201 }202 fwrite(data, 1, data_len, f);203 fclose(f);204 LOG_INF("wrote %s\n", params.out_file.c_str());205 206 llama_backend_free();207 return 0;208}209 