Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
cli-context.cpp679 linesDownload Raw Back to cli
1#include "cli-context.h"2#include "cli-ui.h"3 4#include "arg.h"5#include "base64.hpp"6#include "log.h"7#include "console.h"8 9#define JSON_ASSERT GGML_ASSERT10#include <nlohmann/json.hpp>11 12#include <algorithm>13#include <cctype>14#include <filesystem>15#include <fstream>16#include <map>17#include <set>18 19using json = nlohmann::ordered_json;20 21struct cli_context_impl {22    json messages      = json::array();23    json pending_media = json::array(); // staged multimodal content parts24};25 26cli_context::cli_context(const common_params & params) : params(params), impl(new cli_context_impl()) {}27 28cli_context::~cli_context() {29    shutdown();30}31 32std::atomic<bool> & cli_context::interrupted() {33    static std::atomic<bool> flag = false;34    return flag;35}36 37static bool should_stop() {38    return cli_context::interrupted().load();39}40 41static constexpr size_t FILE_GLOB_MAX_RESULTS = 100;42 43const char * LLAMA_ASCII_LOGO = R"(44▄▄ ▄▄45██ ██46██ ██  ▀▀█▄ ███▄███▄  ▀▀█▄    ▄████ ████▄ ████▄47██ ██ ▄█▀██ ██ ██ ██ ▄█▀██    ██    ██ ██ ██ ██48██ ██ ▀█▄██ ██ ██ ██ ▀█▄██ ██ ▀████ ████▀ ████▀49                                    ██    ██50                                    ▀▀    ▀▀51)";52 53// number of values an arg consumes on the command line54static int arg_num_values(const common_arg & opt) {55    if (opt.value_hint_2 != nullptr) {56        return 2;57    }58    if (opt.value_hint != nullptr) {59        return 1;60    }61    return 0;62}63 64static std::string format_error_message(const json & err) {65    if (err.contains("error") && err.at("error").is_object()) {66        const auto & e = err.at("error");67        if (e.contains("message") && e.at("message").is_string()) {68            return e.at("message").get<std::string>();69        }70    }71    return err.dump();72}73 74// err is the raw response body of a failed request; it may or may not be JSON75static std::string format_error_message(const std::string & err) {76    json parsed = json::parse(err, nullptr, false);77    if (!parsed.is_discarded()) {78        return format_error_message(parsed);79    }80    return err;81}82 83static std::string media_type_from_ext(const std::string & fname) {84    std::string ext = std::filesystem::path(fname).extension().string();85    std::transform(ext.begin(), ext.end(), ext.begin(), [](unsigned char c) { return std::tolower(c); });86    if (ext == ".wav" || ext == ".mp3") {87        return "audio";88    }89    if (ext == ".mp4" || ext == ".avi" || ext == ".mkv" || ext == ".mov" || ext == ".webm") {90        return "video";91    }92    return "image";93}94 95bool cli_context::init() {96    ui::init(params);97 98    std::optional<ui::spinner> spinner;99 100    bool use_external_server = !params.server_base.empty();101    if (use_external_server) {102        std::string base = params.server_base;103        while (!base.empty() && base.back() == '/') {104            base.pop_back();105        }106        client.server_base = base;107 108        spinner.emplace("Connecting to server at " + base);109    } else {110        if (params.model.path.empty() && params.model.url.empty() &&111                params.model.hf_repo.empty() && params.model.docker_repo.empty()) {112            ui::show_error(113                "no model specified",114                "use -m <file.gguf> or -hf <user/repo> to run a local model,\n"115                "or --server-base <url> to connect to a running llama-server"116            );117            return false;118        }119 120        spinner.emplace("\n\nLoading model...");121 122        server.emplace();123        if (!server->start(params)) {124            ui::show_error("server start failed");125            return false;126        }127        if (!server->wait_ready(should_stop)) {128            if (!should_stop()) {129                ui::show_error("the server exited before becoming ready");130            }131            return false;132        }133        client.server_base = server->address();134    }135 136    // for --server-base this is the main availability check; for a spawned137    // server it is a cheap sanity check on top of the ready signal138    auto is_aborted = [this]() {139        return should_stop() || (server && !server->alive());140    };141    bool healthy = false;142    try {143        healthy = client.wait_health(is_aborted);144    } catch (const std::exception & e) {145        client.last_error = e.what();146    }147    if (!healthy) {148        if (!should_stop()) {149            ui::show_error(client.last_error);150        }151        return false;152    }153 154    if (use_external_server) {155        spinner.reset();156        try {157            if (!list_and_ask_models()) {158                return false;159            }160        } catch (const json::parse_error & e) {161            ui::show_error(e.what());162            ui::show_message("This might be caused by an incorrect server-base endpoint URL");163            return false;164        } catch (const std::exception & e) {165            ui::show_error(e.what());166            return false;167        }168 169        // restore the spinner for the next step170        spinner.emplace("Waiting for server...");171    }172 173    fetch_server_props();174 175    if (!params.out_file.empty()) {176        output_file.emplace(params.out_file);177        if (!output_file->is_open()) {178            ui::show_error(string_format("failed to open output file '%s'", params.out_file.c_str()));179            return false;180        }181    }182 183    return true;184}185 186void cli_context::fetch_server_props() {187    try {188        json props = json::parse(client.get("/props"));189        model_name = props.value("model_alias", "");190        if (model_name.empty()) {191            const std::string path = props.value("model_path", "");192            if (!path.empty()) {193                model_name = std::filesystem::path(path).filename().string();194            }195        }196        model_ftype = props.value("model_ftype", "");197        build_info = props.value("build_info", "");198        if (props.contains("modalities") && props.at("modalities").is_object()) {199            const auto & modalities = props.at("modalities");200            has_vision = modalities.value("vision", false);201            has_audio  = modalities.value("audio", false);202            has_video  = modalities.value("video", false);203        }204    } catch (const std::exception & e) {205        // /props can be disabled on remote servers; not fatal206        LOG_DBG("failed to fetch /props: %s\n", e.what());207    }208}209 210bool cli_context::list_and_ask_models() {211    json resp = json::parse(client.get("/v1/models"));212    if (!resp.contains("data") || !resp.at("data").is_array()) {213        throw std::runtime_error("invalid response from /v1/models");214    }215    std::vector<std::string> models;216    std::vector<std::string> models_display;217    for (const auto & m : resp.at("data")) {218        if (!m.contains("id") || !m.at("id").is_string()) {219            continue;220        }221        std::string name = m.at("id").get<std::string>();222        std::string display = name;223        if (m.contains("aliases") && m.at("aliases").is_array()) {224            std::vector<std::string> aliases;225            for (const auto & a : m.at("aliases")) {226                if (a.is_string()) {227                    aliases.push_back(a.get<std::string>());228                }229            }230            if (!aliases.empty()) {231                display += " (" + string_join(aliases, ", ") + ")";232            }233        }234        models.push_back(name);235        models_display.push_back(display);236    }237 238    // only one model: use it without asking239    if (models.size() == 1) {240        model_name = models[0];241        client.model = model_name;242        return true;243    }244 245    std::string message = "\nAvailable models:";246    for (size_t i = 0; i < models_display.size(); ++i) {247        message += "\n  " + std::to_string(i + 1) + ". " + models_display[i];248    }249    message += "\n";250    ui::show_message(message);251    std::string selection;252    while (selection.empty()) {253        if (should_stop()) {254            return false;255        }256        ui::user_turn user_turn;257        selection = user_turn.read_input(false, "Select model by number: ");258        if (selection.empty()) {259            continue;260        }261        try {262            size_t idx = std::stoul(selection);263            if (idx > 0 && idx <= models.size()) {264                model_name = models[idx - 1];265                client.model = model_name;266                ui::show_message("Selected model: " + model_name);267                break;268            }269        } catch (...) {270            // ignore271        }272        ui::show_error("Invalid selection. Please enter a valid number.");273        selection.clear();274        continue;275    }276    return true;277}278 279void cli_context::add_system_prompt() {280    if (!params.system_prompt.empty()) {281        impl->messages.push_back({282            {"role",    "system"},283            {"content", params.system_prompt}284        });285    }286}287 288void cli_context::push_user_message(const std::string & text) {289    json content;290    if (impl->pending_media.empty()) {291        content = text;292    } else {293        // multimodal message: media parts first, then the text294        content = impl->pending_media;295        content.push_back({296            {"type", "text"},297            {"text", text}298        });299        impl->pending_media = json::array();300    }301    impl->messages.push_back({302        {"role",    "user"},303        {"content", content}304    });305}306 307bool cli_context::stage_media_file(const std::string & fname, const std::string & type) {308    std::ifstream file(fname, std::ios::binary);309    if (!file) {310        return false;311    }312    std::string data((std::istreambuf_iterator<char>(file)), std::istreambuf_iterator<char>());313    std::string encoded = base64::encode(data);314 315    if (type == "audio") {316        std::string ext = std::filesystem::path(fname).extension().string();317        std::transform(ext.begin(), ext.end(), ext.begin(), [](unsigned char c) { return std::tolower(c); });318        impl->pending_media.push_back({319            {"type", "input_audio"},320            {"input_audio", {321                {"data",   encoded},322                {"format", ext == ".mp3" ? "mp3" : "wav"}323            }}324        });325    } else if (type == "video") {326        impl->pending_media.push_back({327            {"type", "input_video"},328            {"input_video", {329                {"data", encoded}330            }}331        });332    } else {333        // the server detects the actual image type from the data334        impl->pending_media.push_back({335            {"type", "image_url"},336            {"image_url", {337                {"url", "data:image/unknown;base64," + encoded}338            }}339        });340    }341    return true;342}343 344void cli_context::write_output_file(const std::string & content) {345    if (output_file) {346        (*output_file) << content;347        output_file->flush();348    }349}350 351bool cli_context::generate_completion(generated_content & content_out, cli_timings & timings) {352    json body = {353        {"messages",          impl->messages},354        {"stream",            true},355        // in order to get timings even when we cancel mid-way356        {"timings_per_token", true},357    };358    if (!client.model.empty()) {359        body["model"] = client.model;360    }361 362    bool stream_error = false;363 364    ui::assistant_turn a;365 366    std::string err = client.post_sse("/v1/chat/completions", body.dump(), should_stop, [&](const std::string & payload) {367        json chunk = json::parse(payload, nullptr, false);368        if (chunk.is_discarded()) {369            return;370        }371        if (chunk.contains("error")) {372            stream_error = true;373            ui::show_error(format_error_message(chunk));374            return;375        }376        if (chunk.contains("timings")) {377            const auto & t = chunk.at("timings");378            timings.prompt_per_second    = t.value("prompt_per_second",    0.0);379            timings.predicted_per_second = t.value("predicted_per_second", 0.0);380        }381        if (!chunk.contains("choices") || !chunk.at("choices").is_array() || chunk.at("choices").empty()) {382            return;383        }384        const auto & choice = chunk.at("choices").at(0);385        if (!choice.contains("delta")) {386            return;387        }388        const auto & delta = choice.at("delta");389        if (delta.contains("reasoning_content") && delta.at("reasoning_content").is_string()) {390            const std::string text = delta.at("reasoning_content").get<std::string>();391            if (!text.empty()) {392                content_out.reasoning += text;393                a.push(ui::ASSISTANT_DISPLAY_MODE_REASONING, text);394            }395        }396        if (delta.contains("content") && delta.at("content").is_string()) {397            const std::string text = delta.at("content").get<std::string>();398            if (!text.empty()) {399                content_out.content += text;400                a.push(ui::ASSISTANT_DISPLAY_MODE_CONTENT, text);401            }402        }403    });404 405    cli_context::interrupted().store(false);406 407    if (!err.empty()) {408        ui::show_error(format_error_message(err));409        return false;410    }411    return !stream_error;412}413 414int cli_context::run() {415    add_system_prompt();416 417    std::string modalities = "text";418    if (has_vision) {419        modalities += ", vision";420    }421    if (has_audio) {422        modalities += ", audio";423    }424    if (has_video) {425        modalities += ", video";426    }427 428    std::string banner;429    banner += "\n";430    banner += LLAMA_ASCII_LOGO;431    banner += "\n";432    banner += "build      : " + build_info + "\n";433    banner += "model      : " + model_name + "\n";434    if (!model_ftype.empty()) {435        banner += "ftype      : " + model_ftype + "\n";436    }437    banner += "modalities : " + modalities + "\n";438    if (!params.system_prompt.empty()) {439        banner += "using custom system prompt\n";440    }441    banner += "\n";442    banner += "available commands:\n";443    banner += "  /exit or Ctrl+C     stop or exit\n";444    banner += "  /regen              regenerate the last response\n";445    banner += "  /clear              clear the chat history\n";446    banner += "  /read <file>        add a text file\n";447    banner += "  /glob <pattern>     add text files using globbing pattern\n";448    if (has_vision) {449        banner += "  /image <file>       add an image file\n";450    }451    if (has_audio) {452        banner += "  /audio <file>       add an audio file\n";453    }454    if (has_video) {455        banner += "  /video <file>       add a video file\n";456    }457    banner += "\n";458 459    ui::show_message(banner);460 461    // interactive loop462    std::string cur_msg;463 464    auto add_text_file = [&](const std::string & fname) -> bool {465        std::ifstream file(fname, std::ios::binary);466        if (!file) {467            ui::show_error(string_format("file does not exist or cannot be opened: '%s'", fname.c_str()));468            return false;469        }470        std::string content((std::istreambuf_iterator<char>(file)), std::istreambuf_iterator<char>());471        cur_msg += "--- File: ";472        cur_msg += fname;473        cur_msg += " ---\n";474        cur_msg += content;475        ui::show_message(string_format("Loaded text from '%s'", fname.c_str()));476        return true;477    };478 479    while (true) {480        std::string buffer;481        {482            ui::user_turn user_turn;483 484            if (params.prompt.empty()) {485                buffer = user_turn.read_input(params.multiline_input);486            } else {487                // process input prompt from args488                for (auto & fname : params.image) {489                    if (!stage_media_file(fname, media_type_from_ext(fname))) {490                        ui::show_error(string_format("file does not exist or cannot be opened: '%s'", fname.c_str()));491                        break;492                    }493                    ui::show_message(string_format("Loaded media from '%s'", fname.c_str()));494                }495                buffer = params.prompt;496                user_turn.echo(buffer);497                params.prompt.clear(); // only use it once498            }499        }500 501        if (should_stop()) {502            cli_context::interrupted().store(false);503            break;504        }505 506        // remove trailing newline507        if (!buffer.empty() && buffer.back() == '\n') {508            buffer.pop_back();509        }510 511        // skip empty messages512        if (buffer.empty()) {513            continue;514        }515 516        bool add_user_msg = true;517 518        // process commands519        if (string_starts_with(buffer, "/exit")) {520            break;521        } else if (string_starts_with(buffer, "/regen")) {522            if (impl->messages.size() >= 2) {523                size_t last_idx = impl->messages.size() - 1;524                impl->messages.erase(last_idx);525                add_user_msg = false;526            } else {527                ui::show_error("No message to regenerate.");528                continue;529            }530        } else if (string_starts_with(buffer, "/clear")) {531            impl->messages.clear();532            add_system_prompt();533 534            impl->pending_media = json::array();535            ui::show_message("Chat history cleared.");536            continue;537        } else if (538                (string_starts_with(buffer, "/image ") && has_vision) ||539                (string_starts_with(buffer, "/audio ") && has_audio) ||540                (string_starts_with(buffer, "/video ") && has_video)) {541            std::string type = buffer.substr(1, 5);542            // just in case (bad copy-paste for example), we strip all trailing/leading spaces543            std::string fname = string_strip(buffer.substr(7));544            if (!stage_media_file(fname, type)) {545                ui::show_error(string_format("file does not exist or cannot be opened: '%s'", fname.c_str()));546                continue;547            }548            ui::show_message(string_format("Loaded media from '%s'", fname.c_str()));549            write_output_file(string_format("User: Added media: %s\n", fname.c_str()));550            continue;551        } else if (string_starts_with(buffer, "/read ")) {552            std::string fname = string_strip(buffer.substr(6));553            add_text_file(fname);554            write_output_file(string_format("User: Added text file: %s\n", fname.c_str()));555            continue;556        } else if (string_starts_with(buffer, "/glob ")) {557            std::error_code ec;558            size_t count = 0;559            auto curdir = std::filesystem::current_path();560            std::string pattern = string_strip(buffer.substr(6));561            std::filesystem::path rel_path;562 563            auto startglob = pattern.find_first_of("![*?");564            if (startglob != std::string::npos && startglob != 0) {565                auto endpath = pattern.substr(0, startglob).find_last_of('/');566                if (endpath != std::string::npos) {567                    std::string rel_pattern = pattern.substr(0, endpath);568#if !defined(_WIN32)569                    if (string_starts_with(rel_pattern, '~')) {570                        const char * home = std::getenv("HOME");571                        if (home && home[0]) {572                            rel_pattern = home + rel_pattern.substr(1);573                        }574                    }575#endif576                    rel_path = rel_pattern;577                    pattern.erase(0, endpath + 1);578                    curdir /= rel_path;579                }580            }581 582            for (const auto & entry : std::filesystem::recursive_directory_iterator(curdir,583                    std::filesystem::directory_options::skip_permission_denied, ec)) {584                if (!entry.is_regular_file()) {585                    continue;586                }587 588                std::string rel = std::filesystem::relative(entry.path(), curdir, ec).string();589                if (ec) {590                    ec.clear();591                    continue;592                }593                std::replace(rel.begin(), rel.end(), '\\', '/');594 595                if (!glob_match(pattern, rel)) {596                    continue;597                }598 599                const std::string full_path = (curdir / rel).string();600                if (!add_text_file(full_path)) {601                    continue;602                }603                write_output_file(string_format("User: Added text file: %s\n", full_path.c_str()));604 605                if (++count >= FILE_GLOB_MAX_RESULTS) {606                    ui::show_error(string_format("Maximum number of globbed files allowed (%zu) reached.", FILE_GLOB_MAX_RESULTS));607                    break;608                }609            }610            continue;611        } else {612            // not a command613            cur_msg += buffer;614        }615 616        // generate response617        if (add_user_msg) {618            push_user_message(cur_msg);619            write_output_file(string_format("User:\n%s\n\n", cur_msg.c_str()));620            cur_msg.clear();621        }622 623        cli_timings timings;624        generated_content content;625        generate_completion(content, timings);626 627        json assistant_msg = {628            {"role",    "assistant"},629            {"content", content.content}630        };631        if (!content.reasoning.empty()) {632            assistant_msg["reasoning_content"] = content.reasoning;633        }634        impl->messages.push_back(std::move(assistant_msg));635 636        if (output_file) {637            std::string out_content = "Assistant:\n";638            if (!content.reasoning.empty()) {639                out_content += "[Start thinking]\n\n";640                out_content += content.reasoning;641                out_content += "[End thinking]\n\n";642            }643            out_content += content.content;644            if (!out_content.empty() && out_content.back() != '\n') {645                out_content += "\n";646            }647            out_content += "\n";648            write_output_file(out_content);649        }650 651        if (params.show_timings) {652            ui::show_info(string_format(653                "\n[ Prompt: %.1f t/s | Generation: %.1f t/s ]",654                timings.prompt_per_second,655                timings.predicted_per_second656            ));657        }658 659        if (params.single_turn) {660            break;661        }662    }663 664    ui::show_message("\n\nExiting...");665 666    return 0;667}668 669void cli_context::shutdown() {670    if (server) {671        server->stop();672        server.reset();673    }674    if (output_file) {675        output_file->close();676        output_file.reset();677    }678}679 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai