Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#include "cli-context.h"2#include "cli-ui.h"3 4#include "arg.h"5#include "base64.hpp"6#include "log.h"7#include "console.h"8 9#define JSON_ASSERT GGML_ASSERT10#include <nlohmann/json.hpp>11 12#include <algorithm>13#include <cctype>14#include <filesystem>15#include <fstream>16#include <map>17#include <set>18 19using json = nlohmann::ordered_json;20 21struct cli_context_impl {22 json messages = json::array();23 json pending_media = json::array(); // staged multimodal content parts24};25 26cli_context::cli_context(const common_params & params) : params(params), impl(new cli_context_impl()) {}27 28cli_context::~cli_context() {29 shutdown();30}31 32std::atomic<bool> & cli_context::interrupted() {33 static std::atomic<bool> flag = false;34 return flag;35}36 37static bool should_stop() {38 return cli_context::interrupted().load();39}40 41static constexpr size_t FILE_GLOB_MAX_RESULTS = 100;42 43const char * LLAMA_ASCII_LOGO = R"(44▄▄ ▄▄45██ ██46██ ██ ▀▀█▄ ███▄███▄ ▀▀█▄ ▄████ ████▄ ████▄47██ ██ ▄█▀██ ██ ██ ██ ▄█▀██ ██ ██ ██ ██ ██48██ ██ ▀█▄██ ██ ██ ██ ▀█▄██ ██ ▀████ ████▀ ████▀49 ██ ██50 ▀▀ ▀▀51)";52 53// number of values an arg consumes on the command line54static int arg_num_values(const common_arg & opt) {55 if (opt.value_hint_2 != nullptr) {56 return 2;57 }58 if (opt.value_hint != nullptr) {59 return 1;60 }61 return 0;62}63 64static std::string format_error_message(const json & err) {65 if (err.contains("error") && err.at("error").is_object()) {66 const auto & e = err.at("error");67 if (e.contains("message") && e.at("message").is_string()) {68 return e.at("message").get<std::string>();69 }70 }71 return err.dump();72}73 74// err is the raw response body of a failed request; it may or may not be JSON75static std::string format_error_message(const std::string & err) {76 json parsed = json::parse(err, nullptr, false);77 if (!parsed.is_discarded()) {78 return format_error_message(parsed);79 }80 return err;81}82 83static std::string media_type_from_ext(const std::string & fname) {84 std::string ext = std::filesystem::path(fname).extension().string();85 std::transform(ext.begin(), ext.end(), ext.begin(), [](unsigned char c) { return std::tolower(c); });86 if (ext == ".wav" || ext == ".mp3") {87 return "audio";88 }89 if (ext == ".mp4" || ext == ".avi" || ext == ".mkv" || ext == ".mov" || ext == ".webm") {90 return "video";91 }92 return "image";93}94 95bool cli_context::init() {96 ui::init(params);97 98 std::optional<ui::spinner> spinner;99 100 bool use_external_server = !params.server_base.empty();101 if (use_external_server) {102 std::string base = params.server_base;103 while (!base.empty() && base.back() == '/') {104 base.pop_back();105 }106 client.server_base = base;107 108 spinner.emplace("Connecting to server at " + base);109 } else {110 if (params.model.path.empty() && params.model.url.empty() &&111 params.model.hf_repo.empty() && params.model.docker_repo.empty()) {112 ui::show_error(113 "no model specified",114 "use -m <file.gguf> or -hf <user/repo> to run a local model,\n"115 "or --server-base <url> to connect to a running llama-server"116 );117 return false;118 }119 120 spinner.emplace("\n\nLoading model...");121 122 server.emplace();123 if (!server->start(params)) {124 ui::show_error("server start failed");125 return false;126 }127 if (!server->wait_ready(should_stop)) {128 if (!should_stop()) {129 ui::show_error("the server exited before becoming ready");130 }131 return false;132 }133 client.server_base = server->address();134 }135 136 // for --server-base this is the main availability check; for a spawned137 // server it is a cheap sanity check on top of the ready signal138 auto is_aborted = [this]() {139 return should_stop() || (server && !server->alive());140 };141 bool healthy = false;142 try {143 healthy = client.wait_health(is_aborted);144 } catch (const std::exception & e) {145 client.last_error = e.what();146 }147 if (!healthy) {148 if (!should_stop()) {149 ui::show_error(client.last_error);150 }151 return false;152 }153 154 if (use_external_server) {155 spinner.reset();156 try {157 if (!list_and_ask_models()) {158 return false;159 }160 } catch (const json::parse_error & e) {161 ui::show_error(e.what());162 ui::show_message("This might be caused by an incorrect server-base endpoint URL");163 return false;164 } catch (const std::exception & e) {165 ui::show_error(e.what());166 return false;167 }168 169 // restore the spinner for the next step170 spinner.emplace("Waiting for server...");171 }172 173 fetch_server_props();174 175 if (!params.out_file.empty()) {176 output_file.emplace(params.out_file);177 if (!output_file->is_open()) {178 ui::show_error(string_format("failed to open output file '%s'", params.out_file.c_str()));179 return false;180 }181 }182 183 return true;184}185 186void cli_context::fetch_server_props() {187 try {188 json props = json::parse(client.get("/props"));189 model_name = props.value("model_alias", "");190 if (model_name.empty()) {191 const std::string path = props.value("model_path", "");192 if (!path.empty()) {193 model_name = std::filesystem::path(path).filename().string();194 }195 }196 model_ftype = props.value("model_ftype", "");197 build_info = props.value("build_info", "");198 if (props.contains("modalities") && props.at("modalities").is_object()) {199 const auto & modalities = props.at("modalities");200 has_vision = modalities.value("vision", false);201 has_audio = modalities.value("audio", false);202 has_video = modalities.value("video", false);203 }204 } catch (const std::exception & e) {205 // /props can be disabled on remote servers; not fatal206 LOG_DBG("failed to fetch /props: %s\n", e.what());207 }208}209 210bool cli_context::list_and_ask_models() {211 json resp = json::parse(client.get("/v1/models"));212 if (!resp.contains("data") || !resp.at("data").is_array()) {213 throw std::runtime_error("invalid response from /v1/models");214 }215 std::vector<std::string> models;216 std::vector<std::string> models_display;217 for (const auto & m : resp.at("data")) {218 if (!m.contains("id") || !m.at("id").is_string()) {219 continue;220 }221 std::string name = m.at("id").get<std::string>();222 std::string display = name;223 if (m.contains("aliases") && m.at("aliases").is_array()) {224 std::vector<std::string> aliases;225 for (const auto & a : m.at("aliases")) {226 if (a.is_string()) {227 aliases.push_back(a.get<std::string>());228 }229 }230 if (!aliases.empty()) {231 display += " (" + string_join(aliases, ", ") + ")";232 }233 }234 models.push_back(name);235 models_display.push_back(display);236 }237 238 // only one model: use it without asking239 if (models.size() == 1) {240 model_name = models[0];241 client.model = model_name;242 return true;243 }244 245 std::string message = "\nAvailable models:";246 for (size_t i = 0; i < models_display.size(); ++i) {247 message += "\n " + std::to_string(i + 1) + ". " + models_display[i];248 }249 message += "\n";250 ui::show_message(message);251 std::string selection;252 while (selection.empty()) {253 if (should_stop()) {254 return false;255 }256 ui::user_turn user_turn;257 selection = user_turn.read_input(false, "Select model by number: ");258 if (selection.empty()) {259 continue;260 }261 try {262 size_t idx = std::stoul(selection);263 if (idx > 0 && idx <= models.size()) {264 model_name = models[idx - 1];265 client.model = model_name;266 ui::show_message("Selected model: " + model_name);267 break;268 }269 } catch (...) {270 // ignore271 }272 ui::show_error("Invalid selection. Please enter a valid number.");273 selection.clear();274 continue;275 }276 return true;277}278 279void cli_context::add_system_prompt() {280 if (!params.system_prompt.empty()) {281 impl->messages.push_back({282 {"role", "system"},283 {"content", params.system_prompt}284 });285 }286}287 288void cli_context::push_user_message(const std::string & text) {289 json content;290 if (impl->pending_media.empty()) {291 content = text;292 } else {293 // multimodal message: media parts first, then the text294 content = impl->pending_media;295 content.push_back({296 {"type", "text"},297 {"text", text}298 });299 impl->pending_media = json::array();300 }301 impl->messages.push_back({302 {"role", "user"},303 {"content", content}304 });305}306 307bool cli_context::stage_media_file(const std::string & fname, const std::string & type) {308 std::ifstream file(fname, std::ios::binary);309 if (!file) {310 return false;311 }312 std::string data((std::istreambuf_iterator<char>(file)), std::istreambuf_iterator<char>());313 std::string encoded = base64::encode(data);314 315 if (type == "audio") {316 std::string ext = std::filesystem::path(fname).extension().string();317 std::transform(ext.begin(), ext.end(), ext.begin(), [](unsigned char c) { return std::tolower(c); });318 impl->pending_media.push_back({319 {"type", "input_audio"},320 {"input_audio", {321 {"data", encoded},322 {"format", ext == ".mp3" ? "mp3" : "wav"}323 }}324 });325 } else if (type == "video") {326 impl->pending_media.push_back({327 {"type", "input_video"},328 {"input_video", {329 {"data", encoded}330 }}331 });332 } else {333 // the server detects the actual image type from the data334 impl->pending_media.push_back({335 {"type", "image_url"},336 {"image_url", {337 {"url", "data:image/unknown;base64," + encoded}338 }}339 });340 }341 return true;342}343 344void cli_context::write_output_file(const std::string & content) {345 if (output_file) {346 (*output_file) << content;347 output_file->flush();348 }349}350 351bool cli_context::generate_completion(generated_content & content_out, cli_timings & timings) {352 json body = {353 {"messages", impl->messages},354 {"stream", true},355 // in order to get timings even when we cancel mid-way356 {"timings_per_token", true},357 };358 if (!client.model.empty()) {359 body["model"] = client.model;360 }361 362 bool stream_error = false;363 364 ui::assistant_turn a;365 366 std::string err = client.post_sse("/v1/chat/completions", body.dump(), should_stop, [&](const std::string & payload) {367 json chunk = json::parse(payload, nullptr, false);368 if (chunk.is_discarded()) {369 return;370 }371 if (chunk.contains("error")) {372 stream_error = true;373 ui::show_error(format_error_message(chunk));374 return;375 }376 if (chunk.contains("timings")) {377 const auto & t = chunk.at("timings");378 timings.prompt_per_second = t.value("prompt_per_second", 0.0);379 timings.predicted_per_second = t.value("predicted_per_second", 0.0);380 }381 if (!chunk.contains("choices") || !chunk.at("choices").is_array() || chunk.at("choices").empty()) {382 return;383 }384 const auto & choice = chunk.at("choices").at(0);385 if (!choice.contains("delta")) {386 return;387 }388 const auto & delta = choice.at("delta");389 if (delta.contains("reasoning_content") && delta.at("reasoning_content").is_string()) {390 const std::string text = delta.at("reasoning_content").get<std::string>();391 if (!text.empty()) {392 content_out.reasoning += text;393 a.push(ui::ASSISTANT_DISPLAY_MODE_REASONING, text);394 }395 }396 if (delta.contains("content") && delta.at("content").is_string()) {397 const std::string text = delta.at("content").get<std::string>();398 if (!text.empty()) {399 content_out.content += text;400 a.push(ui::ASSISTANT_DISPLAY_MODE_CONTENT, text);401 }402 }403 });404 405 cli_context::interrupted().store(false);406 407 if (!err.empty()) {408 ui::show_error(format_error_message(err));409 return false;410 }411 return !stream_error;412}413 414int cli_context::run() {415 add_system_prompt();416 417 std::string modalities = "text";418 if (has_vision) {419 modalities += ", vision";420 }421 if (has_audio) {422 modalities += ", audio";423 }424 if (has_video) {425 modalities += ", video";426 }427 428 std::string banner;429 banner += "\n";430 banner += LLAMA_ASCII_LOGO;431 banner += "\n";432 banner += "build : " + build_info + "\n";433 banner += "model : " + model_name + "\n";434 if (!model_ftype.empty()) {435 banner += "ftype : " + model_ftype + "\n";436 }437 banner += "modalities : " + modalities + "\n";438 if (!params.system_prompt.empty()) {439 banner += "using custom system prompt\n";440 }441 banner += "\n";442 banner += "available commands:\n";443 banner += " /exit or Ctrl+C stop or exit\n";444 banner += " /regen regenerate the last response\n";445 banner += " /clear clear the chat history\n";446 banner += " /read <file> add a text file\n";447 banner += " /glob <pattern> add text files using globbing pattern\n";448 if (has_vision) {449 banner += " /image <file> add an image file\n";450 }451 if (has_audio) {452 banner += " /audio <file> add an audio file\n";453 }454 if (has_video) {455 banner += " /video <file> add a video file\n";456 }457 banner += "\n";458 459 ui::show_message(banner);460 461 // interactive loop462 std::string cur_msg;463 464 auto add_text_file = [&](const std::string & fname) -> bool {465 std::ifstream file(fname, std::ios::binary);466 if (!file) {467 ui::show_error(string_format("file does not exist or cannot be opened: '%s'", fname.c_str()));468 return false;469 }470 std::string content((std::istreambuf_iterator<char>(file)), std::istreambuf_iterator<char>());471 cur_msg += "--- File: ";472 cur_msg += fname;473 cur_msg += " ---\n";474 cur_msg += content;475 ui::show_message(string_format("Loaded text from '%s'", fname.c_str()));476 return true;477 };478 479 while (true) {480 std::string buffer;481 {482 ui::user_turn user_turn;483 484 if (params.prompt.empty()) {485 buffer = user_turn.read_input(params.multiline_input);486 } else {487 // process input prompt from args488 for (auto & fname : params.image) {489 if (!stage_media_file(fname, media_type_from_ext(fname))) {490 ui::show_error(string_format("file does not exist or cannot be opened: '%s'", fname.c_str()));491 break;492 }493 ui::show_message(string_format("Loaded media from '%s'", fname.c_str()));494 }495 buffer = params.prompt;496 user_turn.echo(buffer);497 params.prompt.clear(); // only use it once498 }499 }500 501 if (should_stop()) {502 cli_context::interrupted().store(false);503 break;504 }505 506 // remove trailing newline507 if (!buffer.empty() && buffer.back() == '\n') {508 buffer.pop_back();509 }510 511 // skip empty messages512 if (buffer.empty()) {513 continue;514 }515 516 bool add_user_msg = true;517 518 // process commands519 if (string_starts_with(buffer, "/exit")) {520 break;521 } else if (string_starts_with(buffer, "/regen")) {522 if (impl->messages.size() >= 2) {523 size_t last_idx = impl->messages.size() - 1;524 impl->messages.erase(last_idx);525 add_user_msg = false;526 } else {527 ui::show_error("No message to regenerate.");528 continue;529 }530 } else if (string_starts_with(buffer, "/clear")) {531 impl->messages.clear();532 add_system_prompt();533 534 impl->pending_media = json::array();535 ui::show_message("Chat history cleared.");536 continue;537 } else if (538 (string_starts_with(buffer, "/image ") && has_vision) ||539 (string_starts_with(buffer, "/audio ") && has_audio) ||540 (string_starts_with(buffer, "/video ") && has_video)) {541 std::string type = buffer.substr(1, 5);542 // just in case (bad copy-paste for example), we strip all trailing/leading spaces543 std::string fname = string_strip(buffer.substr(7));544 if (!stage_media_file(fname, type)) {545 ui::show_error(string_format("file does not exist or cannot be opened: '%s'", fname.c_str()));546 continue;547 }548 ui::show_message(string_format("Loaded media from '%s'", fname.c_str()));549 write_output_file(string_format("User: Added media: %s\n", fname.c_str()));550 continue;551 } else if (string_starts_with(buffer, "/read ")) {552 std::string fname = string_strip(buffer.substr(6));553 add_text_file(fname);554 write_output_file(string_format("User: Added text file: %s\n", fname.c_str()));555 continue;556 } else if (string_starts_with(buffer, "/glob ")) {557 std::error_code ec;558 size_t count = 0;559 auto curdir = std::filesystem::current_path();560 std::string pattern = string_strip(buffer.substr(6));561 std::filesystem::path rel_path;562 563 auto startglob = pattern.find_first_of("![*?");564 if (startglob != std::string::npos && startglob != 0) {565 auto endpath = pattern.substr(0, startglob).find_last_of('/');566 if (endpath != std::string::npos) {567 std::string rel_pattern = pattern.substr(0, endpath);568#if !defined(_WIN32)569 if (string_starts_with(rel_pattern, '~')) {570 const char * home = std::getenv("HOME");571 if (home && home[0]) {572 rel_pattern = home + rel_pattern.substr(1);573 }574 }575#endif576 rel_path = rel_pattern;577 pattern.erase(0, endpath + 1);578 curdir /= rel_path;579 }580 }581 582 for (const auto & entry : std::filesystem::recursive_directory_iterator(curdir,583 std::filesystem::directory_options::skip_permission_denied, ec)) {584 if (!entry.is_regular_file()) {585 continue;586 }587 588 std::string rel = std::filesystem::relative(entry.path(), curdir, ec).string();589 if (ec) {590 ec.clear();591 continue;592 }593 std::replace(rel.begin(), rel.end(), '\\', '/');594 595 if (!glob_match(pattern, rel)) {596 continue;597 }598 599 const std::string full_path = (curdir / rel).string();600 if (!add_text_file(full_path)) {601 continue;602 }603 write_output_file(string_format("User: Added text file: %s\n", full_path.c_str()));604 605 if (++count >= FILE_GLOB_MAX_RESULTS) {606 ui::show_error(string_format("Maximum number of globbed files allowed (%zu) reached.", FILE_GLOB_MAX_RESULTS));607 break;608 }609 }610 continue;611 } else {612 // not a command613 cur_msg += buffer;614 }615 616 // generate response617 if (add_user_msg) {618 push_user_message(cur_msg);619 write_output_file(string_format("User:\n%s\n\n", cur_msg.c_str()));620 cur_msg.clear();621 }622 623 cli_timings timings;624 generated_content content;625 generate_completion(content, timings);626 627 json assistant_msg = {628 {"role", "assistant"},629 {"content", content.content}630 };631 if (!content.reasoning.empty()) {632 assistant_msg["reasoning_content"] = content.reasoning;633 }634 impl->messages.push_back(std::move(assistant_msg));635 636 if (output_file) {637 std::string out_content = "Assistant:\n";638 if (!content.reasoning.empty()) {639 out_content += "[Start thinking]\n\n";640 out_content += content.reasoning;641 out_content += "[End thinking]\n\n";642 }643 out_content += content.content;644 if (!out_content.empty() && out_content.back() != '\n') {645 out_content += "\n";646 }647 out_content += "\n";648 write_output_file(out_content);649 }650 651 if (params.show_timings) {652 ui::show_info(string_format(653 "\n[ Prompt: %.1f t/s | Generation: %.1f t/s ]",654 timings.prompt_per_second,655 timings.predicted_per_second656 ));657 }658 659 if (params.single_turn) {660 break;661 }662 }663 664 ui::show_message("\n\nExiting...");665 666 return 0;667}668 669void cli_context::shutdown() {670 if (server) {671 server->stop();672 server.reset();673 }674 if (output_file) {675 output_file->close();676 output_file.reset();677 }678}679 