Team Ai
Datasetpublic

echodict/llama.cpp

version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes479downloads
arg.cpp3928 linesDownload Raw Back to common
1#include "arg.h"2 3#include "build-info.h"4#include "chat.h"5#include "common.h"6#include "download.h"7#include "hf-cache.h"8#include "json-schema-to-grammar.h"9#include "log.h"10#include "sampling.h"11#include "speculative.h"12#include "preset.h"13 14// fix problem with std::min and std::max15#if defined(_WIN32)16#define WIN32_LEAN_AND_MEAN17#ifndef NOMINMAX18#   define NOMINMAX19#endif20#include <windows.h>21#endif22 23#define JSON_ASSERT GGML_ASSERT24#include <nlohmann/json.hpp>25 26#include <algorithm>27#include <cinttypes>28#include <climits>29#include <cstdarg>30#include <fstream>31#include <list>32#include <regex>33#include <set>34#include <string>35#include <thread> // for hardware_concurrency36#include <vector>37 38#ifndef __EMSCRIPTEN__39#ifdef __linux__40#include <linux/limits.h>41#elif defined(_WIN32)42#   if !defined(PATH_MAX)43#   define PATH_MAX MAX_PATH44#   endif45#elif defined(_AIX)46#include <sys/limits.h>47#else48#include <sys/syslimits.h>49#endif50#endif51 52#define LLAMA_MAX_URL_LENGTH 2084 // Maximum URL Length in Chrome: 208353 54extern const char * LICENSES[];55 56using json = nlohmann::ordered_json;57using namespace common_arg_utils;58 59static std::initializer_list<enum llama_example> mmproj_examples = {60    LLAMA_EXAMPLE_MTMD,61    LLAMA_EXAMPLE_SERVER,62    LLAMA_EXAMPLE_CLI,63};64 65static std::string read_file(const std::string & fname) {66    std::ifstream file(fname);67    if (!file) {68        throw std::runtime_error(string_format("error: failed to open file '%s'\n", fname.c_str()));69    }70    std::string content((std::istreambuf_iterator<char>(file)), std::istreambuf_iterator<char>());71    file.close();72    return content;73}74 75static const std::vector<common_arg> & get_common_arg_defs() {76    static const std::vector<common_arg> options = [] {77        common_params params;78        auto ctx = common_params_parser_init(params, LLAMA_EXAMPLE_SERVER, nullptr);79        return ctx.options;80    }();81    return options;82}83 84common_arg & common_arg::set_examples(std::initializer_list<enum llama_example> examples) {85    this->examples = examples;86    return *this;87}88 89common_arg & common_arg::set_excludes(std::initializer_list<enum llama_example> excludes) {90    this->excludes = excludes;91    return *this;92}93 94common_arg & common_arg::set_env(const char * env) {95    help = help + "\n(env: " + env + ")";96    this->env = env;97    return *this;98}99 100common_arg & common_arg::set_sparam() {101    is_sparam = true;102    return *this;103}104 105common_arg & common_arg::set_preset_only() {106    is_preset_only = true;107    return *this;108}109 110bool common_arg::in_example(enum llama_example ex) {111    return examples.find(ex) != examples.end();112}113 114bool common_arg::is_exclude(enum llama_example ex) {115    return excludes.find(ex) != excludes.end();116}117 118bool common_arg::get_value_from_env(std::string & output) const {119    if (env == nullptr) return false;120    if (!args_neg.empty()) {121        // for compatibility, we need to check LLAMA_ARG_NO_ env as well122        std::string neg_env = env;123        string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");124        char * neg_value = std::getenv(neg_env.c_str());125        if (neg_value) {126            output = "0"; // falsey127            return true;128        }129    }130    char * value = std::getenv(env);131    if (value) {132        output = value;133        return true;134    }135    return false;136}137 138bool common_arg::has_value_from_env() const {139    if (env != nullptr && !args_neg.empty()) {140        // for compatibility, we need to check LLAMA_ARG_NO_ env as well141        std::string neg_env = env;142        string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");143        if (std::getenv(neg_env.c_str())) {144            return true;145        }146    }147    return env != nullptr && std::getenv(env);148}149 150static std::vector<std::string> break_str_into_lines(std::string input, size_t max_char_per_line) {151    std::vector<std::string> result;152    std::istringstream iss(input);153    std::string line;154    auto add_line = [&](const std::string& l) {155        if (l.length() <= max_char_per_line) {156            result.push_back(l);157        } else {158            std::istringstream line_stream(l);159            std::string word, current_line;160            while (line_stream >> word) {161                if (current_line.length() + !current_line.empty() + word.length() > max_char_per_line) {162                    if (!current_line.empty()) result.push_back(current_line);163                    current_line = word;164                } else {165                    current_line += (!current_line.empty() ? " " : "") + word;166                }167            }168            if (!current_line.empty()) result.push_back(current_line);169        }170    };171    while (std::getline(iss, line)) {172        add_line(line);173    }174    return result;175}176 177std::string common_arg::to_string() const {178    // params for printing to console179    const static int n_leading_spaces = 40;180    const static int n_char_per_line_help = 70; // TODO: detect this based on current console181    std::string leading_spaces(n_leading_spaces, ' ');182 183    std::ostringstream ss;184    auto all_args = get_args(); // also contains args_neg185    for (const auto & arg : all_args) {186        if (arg == all_args.front()) {187            if (all_args.size() == 1) {188                ss << arg;189            } else {190                // first arg is usually abbreviation, we need padding to make it more beautiful191                auto tmp = std::string(arg) + ", ";192                auto spaces = std::string(std::max(0, 7 - (int)tmp.size()), ' ');193                ss << tmp << spaces;194            }195        } else {196            ss << arg << (arg != all_args.back() ? ", " : "");197        }198    }199    if (value_hint) ss << " " << value_hint;200    if (value_hint_2) ss << " " << value_hint_2;201    if (ss.tellp() > n_leading_spaces - 3) {202        // current line is too long, add new line203        ss << "\n" << leading_spaces;204    } else {205        // padding between arg and help, same line206        ss << std::string(leading_spaces.size() - ss.tellp(), ' ');207    }208    const auto help_lines = break_str_into_lines(help, n_char_per_line_help);209    for (const auto & line : help_lines) {210        ss << (&line == &help_lines.front() ? "" : leading_spaces) << line << "\n";211    }212    return ss.str();213}214 215std::vector<std::string> common_arg::get_args() const {216    std::vector<std::string> result;217    for (const auto & arg : args) {218        result.push_back(std::string(arg));219    }220    for (const auto & arg : args_neg) {221        result.push_back(std::string(arg));222    }223    return result;224}225 226std::vector<std::string> common_arg::get_env() const {227    std::vector<std::string> result;228    if (env) {229        result.push_back(std::string(env));230    }231    if (!args_neg.empty() && env) {232        // for compatibility, we need to add LLAMA_ARG_NO_ variant233        std::string neg_env = env;234        string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");235        result.push_back(neg_env);236    }237    return result;238}239 240//241// utils242//243 244// Helper function to parse tensor buffer override strings245static void parse_tensor_buffer_overrides(const std::string & value, std::vector<llama_model_tensor_buft_override> & overrides) {246    std::map<std::string, ggml_backend_buffer_type_t> buft_list;247    for (size_t i = 0; i < ggml_backend_dev_count(); ++i) {248        auto * dev = ggml_backend_dev_get(i);249        auto * buft = ggml_backend_dev_buffer_type(dev);250        if (buft) {251            buft_list[ggml_backend_buft_name(buft)] = buft;252        }253    }254 255    for (const auto & override : string_split<std::string>(value, ',')) {256        std::string::size_type pos = override.find('=');257        if (pos == std::string::npos) {258            throw std::invalid_argument("invalid value");259        }260        std::string tensor_name = override.substr(0, pos);261        std::string buffer_type = override.substr(pos + 1);262 263        if (buft_list.find(buffer_type) == buft_list.end()) {264            printf("Available buffer types:\n");265            for (const auto & it : buft_list) {266                printf("  %s\n", ggml_backend_buft_name(it.second));267            }268            throw std::invalid_argument("unknown buffer type");269        }270        // keep strings alive and avoid leaking memory by storing them in a static vector271        static std::list<std::string> buft_overrides;272        buft_overrides.push_back(tensor_name);273        overrides.push_back({buft_overrides.back().c_str(), buft_list.at(buffer_type)});274    }275}276 277static std::string clean_file_name(const std::string & fname) {278    std::string clean_fname = fname;279    string_replace_all(clean_fname, "\\", "_");280    string_replace_all(clean_fname, "/", "_");281    return clean_fname;282}283 284static bool common_params_handle_remote_preset(common_params & params, llama_example ex) {285    GGML_ASSERT(!params.model.hf_repo.empty());286 287    // the returned hf_repo is without tag288    auto [hf_repo, hf_tag] = common_download_split_repo_tag(params.model.hf_repo);289 290    // "latest" tag (default if not specified) is translated to "default" preset291    if (hf_tag == "latest") {292        hf_tag = "default";293    }294 295    std::string model_endpoint = common_get_model_endpoint();296    auto preset_url = model_endpoint + hf_repo + "/resolve/main/preset.ini";297 298    // prepare local path for caching299    auto preset_fname = clean_file_name(hf_repo + "_preset.ini");300    auto preset_path = fs_get_cache_file(preset_fname);301    common_download_opts opts;302    opts.bearer_token = params.hf_token;303    opts.offline = params.offline;304    const int status = common_download_file_single(preset_url, preset_path, opts);305    const bool has_preset = status >= 200 && status < 400;306 307    // remote preset is optional, so we don't error out if not found308    if (has_preset) {309        LOG_INF("applying remote preset from %s\n", preset_url.c_str());310        common_preset_context ctx(ex, /* only_remote_allowed */ true);311        common_preset global;312        auto remote_presets = ctx.load_from_ini(preset_path, global);313        remote_presets = ctx.cascade(global, remote_presets);314        if (remote_presets.find(hf_tag) != remote_presets.end()) {315            common_preset preset = remote_presets.at(hf_tag);316            LOG_INF("\n%s", preset.to_ini().c_str()); // to_ini already added trailing newline317            preset.apply_to_params(params);318        } else {319            throw std::runtime_error("Remote preset.ini does not contain [" + std::string(hf_tag) + "] section");320        }321    } else {322        LOG_INF("%s", "no remote preset found, skipping\n");323    }324 325    return has_preset;326}327 328struct handle_model_result {329    bool found_mmproj = false;330    common_params_model mmproj;331};332 333static handle_model_result common_params_handle_model(struct common_params_model & model,334                                                      const std::string          & bearer_token,335                                                      bool                         offline) {336    handle_model_result result;337 338    if (!model.docker_repo.empty()) {339        model.path = common_docker_resolve_model(model.docker_repo);340        model.name = model.docker_repo;341    } else if (!model.hf_repo.empty()) {342        // If -m was used with -hf, treat the model "path" as the hf_file to download343        if (model.hf_file.empty() && !model.path.empty()) {344            model.hf_file = model.path;345            model.path = "";346        }347        common_download_opts opts;348        opts.bearer_token = bearer_token;349        opts.offline = offline;350        auto download_result = common_download_model(model, opts, true);351 352        if (download_result.model_path.empty()) {353            LOG_ERR("error: failed to download model from Hugging Face\n");354            exit(1);355        }356 357        model.name = model.hf_repo;358        model.path = download_result.model_path;359 360        if (!download_result.mmproj_path.empty()) {361            result.found_mmproj = true;362            result.mmproj.path  = download_result.mmproj_path;363        }364    } else if (!model.url.empty()) {365        if (model.path.empty()) {366            auto f = string_split<std::string>(model.url, '#').front();367            f = string_split<std::string>(f, '?').front();368            model.path = fs_get_cache_file(string_split<std::string>(f, '/').back());369        }370 371        common_download_opts opts;372        opts.bearer_token = bearer_token;373        opts.offline = offline;374        auto download_result = common_download_model(model, opts);375        if (download_result.model_path.empty()) {376            LOG_ERR("error: failed to download model from %s\n", model.url.c_str());377            exit(1);378        }379    }380 381    return result;382}383 384const std::vector<ggml_type> kv_cache_types = {385    GGML_TYPE_F32,386    GGML_TYPE_F16,387    GGML_TYPE_BF16,388    GGML_TYPE_Q8_0,389    GGML_TYPE_Q4_0,390    GGML_TYPE_Q4_1,391    GGML_TYPE_IQ4_NL,392    GGML_TYPE_Q5_0,393    GGML_TYPE_Q5_1,394};395 396static ggml_type kv_cache_type_from_str(const std::string & s) {397    for (const auto & type : kv_cache_types) {398        if (ggml_type_name(type) == s) {399            return type;400        }401    }402    throw std::runtime_error("Unsupported cache type: " + s);403}404 405static std::string get_all_kv_cache_types() {406    std::ostringstream msg;407    for (const auto & type : kv_cache_types) {408        msg << ggml_type_name(type) << (&type == &kv_cache_types.back() ? "" : ", ");409    }410    return msg.str();411}412 413static bool parse_bool_value(const std::string & value) {414    if (is_truthy(value)) {415        return true;416    } else if (is_falsey(value)) {417        return false;418    } else {419        throw std::invalid_argument("invalid boolean value");420    }421}422 423//424// CLI argument parsing functions425//426 427static bool common_params_parse_ex(int argc, char ** argv, common_params_context & ctx_arg) {428    common_params & params = ctx_arg.params;429 430    // setup log directly from params.verbosity: see tools/cli/cli.cpp431    common_log_set_verbosity_thold(params.verbosity);432 433    std::unordered_map<std::string, std::pair<common_arg *, bool>> arg_to_options;434    for (auto & opt : ctx_arg.options) {435        for (const auto & arg : opt.args) {436            arg_to_options[arg] = {&opt, /* is_positive */ true};437        }438        for (const auto & arg : opt.args_neg) {439            arg_to_options[arg] = {&opt, /* is_positive */ false};440        }441    }442 443    // handle environment variables444    for (auto & opt : ctx_arg.options) {445        std::string value;446        if (opt.get_value_from_env(value)) {447            try {448                if (opt.handler_void && is_truthy(value)) {449                    opt.handler_void(params);450                }451                if (opt.handler_int) {452                    opt.handler_int(params, std::stoi(value));453                }454                if (opt.handler_bool) {455                    opt.handler_bool(params, parse_bool_value(value));456                }457                if (opt.handler_string) {458                    opt.handler_string(params, value);459                    continue;460                }461            } catch (std::exception & e) {462                throw std::invalid_argument(string_format(463                    "error while handling environment variable \"%s\": %s\n\n", opt.env, e.what()));464            }465        }466    }467 468    // handle command line arguments469    auto check_arg = [&](int i) {470        if (i+1 >= argc) {471            throw std::invalid_argument("expected value for argument");472        }473    };474 475    auto parse_cli_args = [&]() {476        std::set<std::string> seen_args;477 478        for (int i = 1; i < argc; i++) {479            const std::string arg_prefix = "--";480 481            std::string arg = argv[i];482            if (arg.compare(0, arg_prefix.size(), arg_prefix) == 0) {483                std::replace(arg.begin(), arg.end(), '_', '-');484            }485            if (arg_to_options.find(arg) == arg_to_options.end()) {486                throw std::invalid_argument(string_format("error: invalid argument: %s", arg.c_str()));487            }488            if (!seen_args.insert(arg).second) {489                LOG_WRN("DEPRECATED: argument '%s' specified multiple times, use comma-separated values instead (only last value will be used)\n", arg.c_str());490            }491            auto & tmp = arg_to_options[arg];492            auto opt = *tmp.first;493            bool is_positive = tmp.second;494            if (opt.has_value_from_env()) {495                fprintf(stderr, "warn: %s environment variable is set, but will be overwritten by command line argument %s\n", opt.env, arg.c_str());496            }497            try {498                if (opt.handler_void) {499                    opt.handler_void(params);500                    continue;501                }502                if (opt.handler_bool) {503                    opt.handler_bool(params, is_positive);504                    continue;505                }506 507                // arg with single value508                check_arg(i);509                std::string val = argv[++i];510                if (opt.handler_int) {511                    opt.handler_int(params, std::stoi(val));512                    continue;513                }514                if (opt.handler_string) {515                    opt.handler_string(params, val);516                    continue;517                }518 519                // arg with 2 values520                check_arg(i);521                std::string val2 = argv[++i];522                if (opt.handler_str_str) {523                    opt.handler_str_str(params, val, val2);524                    continue;525                }526            } catch (std::exception & e) {527                throw std::invalid_argument(string_format(528                    "error while handling argument \"%s\": %s\n\n"529                    "usage:\n%s\n\nto show complete usage, run with -h",530                    arg.c_str(), e.what(), opt.to_string().c_str()));531            }532        }533    };534 535    // parse the first time to get -hf option (used for remote preset)536    parse_cli_args();537 538    // TODO: Remove later539    try {540        hf_cache::migrate_old_cache_to_hf_cache(params.hf_token, params.offline);541    } catch (const std::exception & e) {542        LOG_WRN("HF cache migration failed: %s\n", e.what());543    }544    // export_graph_ops loads only metadata545    const bool skip_model_download = ctx_arg.ex == LLAMA_EXAMPLE_EXPORT_GRAPH_OPS;546 547    // maybe handle remote preset548    if (!params.model.hf_repo.empty() && !skip_model_download) {549        std::string cli_hf_repo = params.model.hf_repo;550        bool has_preset = common_params_handle_remote_preset(params, ctx_arg.ex);551 552        // special case: if hf_repo explicitly set by preset, we need to preserve it (ignore CLI value)553        // this is useful when we have one HF repo pointing to other HF repos (one model - multiple GGUFs)554        std::string preset_hf_repo = params.model.hf_repo;555        bool preset_has_hf_repo = preset_hf_repo != cli_hf_repo;556 557        if (has_preset) {558            // re-parse CLI args to override preset values559            parse_cli_args();560        }561 562        // preserve hf_repo from preset if needed563        if (preset_has_hf_repo) {564            params.model.hf_repo = preset_hf_repo;565        }566    }567 568    postprocess_cpu_params(params.cpuparams,       nullptr);569    postprocess_cpu_params(params.cpuparams_batch, &params.cpuparams);570 571    postprocess_cpu_params(params.speculative.cpuparams,       &params.cpuparams);572    postprocess_cpu_params(params.speculative.cpuparams_batch, &params.cpuparams_batch);573 574    if (params.prompt_cache_all && (params.interactive || params.interactive_first)) {575        throw std::invalid_argument("error: --prompt-cache-all not supported in interactive mode yet\n");576    }577 578    // handle model and download579    if (!skip_model_download) {580        auto res = common_params_handle_model(params.model, params.hf_token, params.offline);581        if (params.no_mmproj) {582            params.mmproj = {};583        } else if (res.found_mmproj && params.mmproj.path.empty() && params.mmproj.url.empty()) {584            // optionally, handle mmproj model when -hf is specified585            params.mmproj = res.mmproj;586        }587        // only download mmproj if the current example is using it588        for (const auto & ex : mmproj_examples) {589            if (ctx_arg.ex == ex) {590                common_params_handle_model(params.mmproj,    params.hf_token, params.offline);591                break;592            }593        }594        common_params_handle_model(params.speculative.mparams_dft, params.hf_token, params.offline);595        common_params_handle_model(params.vocoder.model,           params.hf_token, params.offline);596    }597 598    // model is required (except for server)599    // TODO @ngxson : maybe show a list of available models in CLI in this case600    if (params.model.path.empty() && ctx_arg.ex != LLAMA_EXAMPLE_SERVER && !skip_model_download && !params.usage && !params.completion) {601        throw std::invalid_argument("error: --model is required\n");602    }603 604    if (params.escape) {605        string_process_escapes(params.prompt);606        string_process_escapes(params.input_prefix);607        string_process_escapes(params.input_suffix);608        for (auto & antiprompt : params.antiprompt) {609            string_process_escapes(antiprompt);610        }611        for (auto & seq_breaker : params.sampling.dry_sequence_breakers) {612            string_process_escapes(seq_breaker);613        }614        for (auto & pair : params.speculative.replacements) {615            string_process_escapes(pair.first);616            string_process_escapes(pair.second);617        }618    }619 620    if (!params.kv_overrides.empty()) {621        params.kv_overrides.emplace_back();622        params.kv_overrides.back().key[0] = 0;623    }624 625    // pad tensor_buft_overrides for llama_params_fit:626    const size_t ntbo = llama_max_tensor_buft_overrides();627    while (params.tensor_buft_overrides.size() < ntbo) {628        params.tensor_buft_overrides.push_back({nullptr, nullptr});629    }630 631    if (!params.speculative.tensor_buft_overrides.empty()) {632        params.speculative.tensor_buft_overrides.push_back({nullptr, nullptr});633    }634 635    if (!params.chat_template.empty() && !common_chat_verify_template(params.chat_template, params.use_jinja)) {636        throw std::runtime_error(string_format(637            "error: the supplied chat template is not supported: %s%s\n",638            params.chat_template.c_str(),639            params.use_jinja ? "" : "\nnote: llama.cpp was started without --jinja, we only support commonly used templates"640        ));641    }642 643    return true;644}645 646static void common_params_print_usage(common_params_context & ctx_arg) {647    auto print_options = [](std::vector<common_arg *> & options) {648        for (common_arg * opt : options) {649            printf("%s", opt->to_string().c_str());650        }651    };652 653    std::vector<common_arg *> common_options;654    std::vector<common_arg *> sparam_options;655    std::vector<common_arg *> specific_options;656    for (auto & opt : ctx_arg.options) {657        // in case multiple LLAMA_EXAMPLE_* are set, we prioritize the LLAMA_EXAMPLE_* matching current example658        if (opt.is_sparam) {659            sparam_options.push_back(&opt);660        } else if (opt.in_example(ctx_arg.ex)) {661            specific_options.push_back(&opt);662        } else {663            common_options.push_back(&opt);664        }665    }666    printf("----- common params -----\n\n");667    print_options(common_options);668    printf("\n\n----- sampling params -----\n\n");669    print_options(sparam_options);670    // TODO: maybe convert enum llama_example to string671    printf("\n\n----- example-specific params -----\n\n");672    print_options(specific_options);673}674 675static void common_params_print_completion(common_params_context & ctx_arg) {676    std::vector<common_arg *> common_options;677    std::vector<common_arg *> sparam_options;678    std::vector<common_arg *> specific_options;679 680    for (auto & opt : ctx_arg.options) {681        if (opt.is_sparam) {682            sparam_options.push_back(&opt);683        } else if (opt.in_example(ctx_arg.ex)) {684            specific_options.push_back(&opt);685        } else {686            common_options.push_back(&opt);687        }688    }689 690    printf("_llama_completions() {\n");691    printf("    local cur prev opts\n");692    printf("    COMPREPLY=()\n");693    printf("    cur=\"${COMP_WORDS[COMP_CWORD]}\"\n");694    printf("    prev=\"${COMP_WORDS[COMP_CWORD-1]}\"\n\n");695 696    printf("    opts=\"");697    auto print_options = [](const std::vector<common_arg *> & options) {698        for (const common_arg * opt : options) {699            for (const char * arg : opt->args) {700                printf("%s ", arg);701            }702        }703    };704 705    print_options(common_options);706    print_options(sparam_options);707    print_options(specific_options);708    printf("\"\n\n");709 710    printf("    case \"$prev\" in\n");711    printf("        --model|-m)\n");712    printf("            COMPREPLY=( $(compgen -f -X '!*.gguf' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");713    printf("            return 0\n");714    printf("            ;;\n");715    printf("        --grammar-file)\n");716    printf("            COMPREPLY=( $(compgen -f -X '!*.gbnf' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");717    printf("            return 0\n");718    printf("            ;;\n");719    printf("        --chat-template-file)\n");720    printf("            COMPREPLY=( $(compgen -f -X '!*.jinja' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");721    printf("            return 0\n");722    printf("            ;;\n");723    printf("        *)\n");724    printf("            COMPREPLY=( $(compgen -W \"${opts}\" -- \"$cur\") )\n");725    printf("            return 0\n");726    printf("            ;;\n");727    printf("    esac\n");728    printf("}\n\n");729 730    std::set<std::string> executables = {731        "llama-batched",732        "llama-batched-bench",733        "llama-bench",734        "llama-cli",735        "llama-completion",736        "llama-convert-llama2c-to-ggml",737        "llama-cvector-generator",738        "llama-debug",739        "llama-diffusion-cli",740        "llama-embedding",741        "llama-eval-callback",742        "llama-export-lora",743        "llama-finetune",744        "llama-fit-params",745        "llama-gemma3-cli",746        "llama-gen-docs",747        "llama-gguf",748        "llama-gguf-hash",749        "llama-gguf-split",750        "llama-idle",751        "llama-imatrix",752        "llama-llava-cli",753        "llama-lookahead",754        "llama-lookup",755        "llama-lookup-create",756        "llama-lookup-merge",757        "llama-lookup-stats",758        "llama-minicpmv-cli",759        "llama-mtmd-cli",760        "llama-parallel",761        "llama-passkey",762        "llama-perplexity",763        "llama-q8dot",764        "llama-quantize",765        "llama-qwen2vl-cli",766        "llama-retrieval",767        "llama-save-load-state",768        "llama-server",769        "llama-simple",770        "llama-simple-chat",771        "llama-speculative",772        "llama-speculative-simple",773        "llama-tokenize",774        "llama-tts",775        "llama-vdot"776    };777 778    for (const auto& exe : executables) {779        printf("complete -F _llama_completions %s\n", exe.c_str());780    }781}782 783static std::vector<ggml_backend_dev_t> parse_device_list(const std::string & value) {784    std::vector<ggml_backend_dev_t> devices;785    auto dev_names = string_split<std::string>(value, ',');786    if (dev_names.empty()) {787        throw std::invalid_argument("no devices specified");788    }789    if (dev_names.size() == 1 && dev_names[0] == "none") {790        devices.push_back(nullptr);791    } else {792        for (const auto & device : dev_names) {793            auto * dev = ggml_backend_dev_by_name(device.c_str());794            if (!dev || ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_CPU) {795                throw std::invalid_argument(string_format("invalid device: %s", device.c_str()));796            }797            devices.push_back(dev);798        }799        devices.push_back(nullptr);800    }801    return devices;802}803 804static void add_rpc_devices(const std::string & servers) {805    auto rpc_servers = string_split<std::string>(servers, ',');806    if (rpc_servers.empty()) {807        throw std::invalid_argument("no RPC servers specified");808    }809    ggml_backend_reg_t rpc_reg = ggml_backend_reg_by_name("RPC");810    if (!rpc_reg) {811        throw std::invalid_argument("failed to find RPC backend");812    }813    typedef ggml_backend_reg_t (*ggml_backend_rpc_add_server_t)(const char * endpoint);814    ggml_backend_rpc_add_server_t ggml_backend_rpc_add_server_fn = (ggml_backend_rpc_add_server_t) ggml_backend_reg_get_proc_address(rpc_reg, "ggml_backend_rpc_add_server");815    if (!ggml_backend_rpc_add_server_fn) {816        throw std::invalid_argument("failed to find RPC add server function");817    }818    for (const auto & server : rpc_servers) {819        auto reg = ggml_backend_rpc_add_server_fn(server.c_str());820        ggml_backend_register(reg);821    }822}823 824bool common_params_to_map(int argc, char ** argv, llama_example ex, std::map<common_arg, std::string> & out_map) {825    common_params dummy_params;826    common_params_context ctx_arg = common_params_parser_init(dummy_params, ex, nullptr);827 828    std::unordered_map<std::string, common_arg *> arg_to_options;829    for (auto & opt : ctx_arg.options) {830        for (const auto & arg : opt.args) {831            arg_to_options[arg] = &opt;832        }833        for (const auto & arg : opt.args_neg) {834            arg_to_options[arg] = &opt;835        }836    }837 838    // TODO @ngxson : find a way to deduplicate this code839 840    // handle command line arguments841    auto check_arg = [&](int i) {842        if (i+1 >= argc) {843            throw std::invalid_argument("expected value for argument");844        }845    };846 847    std::set<std::string> seen_args;848 849    for (int i = 1; i < argc; i++) {850        const std::string arg_prefix = "--";851 852        std::string arg = argv[i];853        if (arg.compare(0, arg_prefix.size(), arg_prefix) == 0) {854            std::replace(arg.begin(), arg.end(), '_', '-');855        }856        if (arg_to_options.find(arg) == arg_to_options.end()) {857            throw std::invalid_argument(string_format("error: invalid argument: %s", arg.c_str()));858        }859        if (!seen_args.insert(arg).second) {860            LOG_WRN("DEPRECATED: argument '%s' specified multiple times, use comma-separated values instead (only last value will be used)\n", arg.c_str());861        }862        auto opt = *arg_to_options[arg];863        std::string val;864        if (opt.value_hint == nullptr && opt.value_hint_2 == nullptr) {865            // bool arg (need to reverse the meaning for negative args)866            bool is_neg = std::find(opt.args_neg.begin(), opt.args_neg.end(), arg) != opt.args_neg.end();867            val = is_neg ? "0" : "1";868        }869        if (opt.value_hint != nullptr) {870            // arg with single value871            check_arg(i);872            val = argv[++i];873        }874        if (opt.value_hint_2 != nullptr) {875            // TODO: support arg with 2 values876            throw std::invalid_argument("error: argument with 2 values is not yet supported\n");877        }878        out_map[opt] = val;879    }880 881    return true;882}883 884bool common_params_parse(int argc, char ** argv, common_params & params, llama_example ex, void(*print_usage)(int, char **)) {885    auto ctx_arg = common_params_parser_init(params, ex, print_usage);886    const common_params params_org = ctx_arg.params; // the example can modify the default params887 888    try {889        if (!common_params_parse_ex(argc, argv, ctx_arg)) {890            ctx_arg.params = params_org;891            return false;892        }893        if (ctx_arg.params.usage) {894            common_params_print_usage(ctx_arg);895            if (ctx_arg.print_usage) {896                ctx_arg.print_usage(argc, argv);897            }898            exit(0);899        }900        if (ctx_arg.params.completion) {901            common_params_print_completion(ctx_arg);902            exit(0);903        }904        params.lr.init();905    } catch (const std::invalid_argument & ex) {906        fprintf(stderr, "%s\n", ex.what());907        ctx_arg.params = params_org;908        return false;909    } catch (std::exception & ex) {910        fprintf(stderr, "%s\n", ex.what());911        exit(1); // for other exceptions, we exit with status code 1912    }913 914    return true;915}916 917static std::string list_builtin_chat_templates() {918    std::vector<const char *> supported_tmpl;919    int32_t res = llama_chat_builtin_templates(nullptr, 0);920    supported_tmpl.resize(res);921    res = llama_chat_builtin_templates(supported_tmpl.data(), supported_tmpl.size());922    std::ostringstream msg;923    for (auto & tmpl : supported_tmpl) {924        msg << tmpl << (&tmpl == &supported_tmpl.back() ? "" : ", ");925    }926    return msg.str();927}928 929bool common_arg_utils::is_truthy(const std::string & value) {930    return value == "on" || value == "enabled" || value == "true" || value == "1";931}932 933bool common_arg_utils::is_falsey(const std::string & value) {934    return value == "off" || value == "disabled" || value == "false" || value == "0";935}936 937bool common_arg_utils::is_autoy(const std::string & value) {938    return value == "auto" || value == "-1";939}940 941// Simple CSV parser that handles quoted fields and escaped quotes942// example:943//    input:  value1,"value, with, commas","value with ""escaped"" quotes",value4944//    output: [value1] [value, with, commas] [value with "escaped" quotes] [value4]945static std::vector<std::string> parse_csv_row(const std::string& input) {946    std::vector<std::string> fields;947    std::string field;948    bool in_quotes = false;949 950    for (size_t i = 0; i < input.length(); ++i) {951        char ch = input[i];952 953        if (ch == '"') {954            if (!in_quotes) {955                // start of quoted field (only valid if at beginning of field)956                if (!field.empty()) {957                    // quote appeared in middle of unquoted field, treat as literal958                    field += '"';959                } else {960                    in_quotes = true; // start961                }962            } else {963                if (i + 1 < input.length() && input[i + 1] == '"') {964                    // escaped quote: ""965                    field += '"';966                    ++i; // skip the next quote967                } else {968                    in_quotes = false; // end969                }970            }971        } else if (ch == ',') {972            if (in_quotes) {973                field += ',';974            } else {975                fields.push_back(std::move(field));976                field.clear();977            }978        } else {979            field += ch;980        }981    }982 983    // Add the last field984    fields.push_back(std::move(field));985 986    return fields;987}988 989common_params_context common_params_parser_init(common_params & params, llama_example ex, void(*print_usage)(int, char **)) {990    // per-example default params991    // we define here to make sure it's included in llama-gen-docs992    if (ex == LLAMA_EXAMPLE_COMPLETION) {993        params.use_jinja = false;   // disable jinja by default994 995    } else if (ex == LLAMA_EXAMPLE_MTMD) {996        params.use_jinja = false;   // disable jinja by default997        params.sampling.temp = 0.2; // lower temp by default for better quality998 999    } else if (ex == LLAMA_EXAMPLE_SERVER) {1000        params.n_parallel = -1;     // auto by default1001    }1002 1003    params.use_color = tty_can_use_colors();1004 1005    // load dynamic backends1006    ggml_backend_load_all();1007 1008    common_params_context ctx_arg(params);1009    ctx_arg.print_usage = print_usage;1010    ctx_arg.ex          = ex;1011 1012    std::string sampler_type_chars;1013    std::string sampler_type_names;1014    for (const auto & sampler : params.sampling.samplers) {1015        sampler_type_chars += common_sampler_type_to_chr(sampler);1016        sampler_type_names += common_sampler_type_to_str(sampler) + ";";1017    }1018    if (!sampler_type_names.empty()) {1019        sampler_type_names.pop_back(); // remove last semicolon1020    }1021 1022 1023    /**1024     * filter options by example1025     * rules:1026     * - all examples inherit options from LLAMA_EXAMPLE_COMMON1027     * - if LLAMA_EXAMPLE_* is set (other than COMMON), we only show the option in the corresponding example1028     * - if both {LLAMA_EXAMPLE_COMMON, LLAMA_EXAMPLE_*,} are set, we will prioritize the LLAMA_EXAMPLE_* matching current example1029     */1030    auto add_opt = [&](common_arg arg) {1031        if ((arg.in_example(ex) || arg.in_example(LLAMA_EXAMPLE_COMMON)) && !arg.is_exclude(ex)) {1032            ctx_arg.options.push_back(std::move(arg));1033        }1034    };1035 1036 1037    add_opt(common_arg(1038        {"-h", "--help", "--usage"},1039        "print usage and exit",1040        [](common_params & params) {1041            params.usage = true;1042        }1043    ));1044    add_opt(common_arg(1045        {"--version"},1046        "show version and build info",1047        [](common_params &) {1048            fprintf(stderr, "version: %d (%s)\n", llama_build_number(), llama_commit());1049            fprintf(stderr, "built with %s for %s\n", llama_compiler(), llama_build_target());1050            exit(0);1051        }1052    ));1053    add_opt(common_arg(1054        {"--license"},1055        "show source code license and dependencies",1056        [](common_params &) {1057            for (int i = 0; LICENSES[i]; ++i) {1058                printf("%s\n", LICENSES[i]);1059            }1060            exit(0);1061        }1062    ));1063    add_opt(common_arg(1064        {"-cl", "--cache-list"},1065        "show list of models in cache",1066        [](common_params &) {1067            auto models = common_list_cached_models();1068            printf("number of models in cache: %zu\n", models.size());1069            for (size_t i = 0; i < models.size(); i++) {1070                printf("%4zu. %s\n", i + 1, models[i].to_string().c_str());1071            }1072            exit(0);1073        }1074    ));1075    add_opt(common_arg(1076        {"--completion-bash"},1077        "print source-able bash completion script for llama.cpp",1078        [](common_params & params) {1079            params.completion = true;1080        }1081    ));1082    add_opt(common_arg(1083        {"--verbose-prompt"},1084        string_format("print a verbose prompt before generation (default: %s)", params.verbose_prompt ? "true" : "false"),1085        [](common_params & params) {1086            params.verbose_prompt = true;1087        }1088    ).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_EMBEDDING, LLAMA_EXAMPLE_RETRIEVAL}));1089    add_opt(common_arg(1090        {"--display-prompt"},1091        {"--no-display-prompt"},1092        string_format("whether to print prompt at generation (default: %s)", params.display_prompt ? "true" : "false"),1093        [](common_params & params, bool value) {1094            params.display_prompt = value;1095        }1096    ).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI}));1097    add_opt(common_arg(1098        {"-co", "--color"}, "[on|off|auto]",1099        "Colorize output to distinguish prompt and user input from generations ('on', 'off', or 'auto', default: 'auto')\n"1100        "'auto' enables colors when output is to a terminal",1101        [](common_params & params, const std::string & value) {1102            if (is_truthy(value)) {1103                params.use_color = true;1104            } else if (is_falsey(value)) {1105                params.use_color = false;1106            } else if (is_autoy(value)) {1107                params.use_color = tty_can_use_colors();1108            } else {1109                throw std::invalid_argument(1110                    string_format("error: unknown value for --color: '%s'\n", value.c_str()));1111            }1112        }1113    ).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_LOOKUP}));1114    add_opt(common_arg(1115        {"-t", "--threads"}, "N",1116        string_format("number of CPU threads to use during generation (default: %d)", params.cpuparams.n_threads),1117        [](common_params & params, int value) {1118            params.cpuparams.n_threads = value;1119            if (params.cpuparams.n_threads <= 0) {1120                params.cpuparams.n_threads = std::thread::hardware_concurrency();1121            }1122        }1123    ).set_env("LLAMA_ARG_THREADS"));1124    add_opt(common_arg(1125        {"-tb", "--threads-batch"}, "N",1126        "number of threads to use during batch and prompt processing (default: same as --threads)",1127        [](common_params & params, int value) {1128            params.cpuparams_batch.n_threads = value;1129            if (params.cpuparams_batch.n_threads <= 0) {1130                params.cpuparams_batch.n_threads = std::thread::hardware_concurrency();1131            }1132        }1133    ));1134    add_opt(common_arg(1135        {"-C", "--cpu-mask"}, "M",1136        "CPU affinity mask: arbitrarily long hex. Complements cpu-range (default: \"\")",1137        [](common_params & params, const std::string & mask) {1138            params.cpuparams.mask_valid = true;1139            if (!parse_cpu_mask(mask, params.cpuparams.cpumask)) {1140                throw std::invalid_argument("invalid cpumask");1141            }1142        }1143    ));1144    add_opt(common_arg(1145        {"-Cr", "--cpu-range"}, "lo-hi",1146        "range of CPUs for affinity. Complements --cpu-mask",1147        [](common_params & params, const std::string & range) {1148            params.cpuparams.mask_valid = true;1149            if (!parse_cpu_range(range, params.cpuparams.cpumask)) {1150                throw std::invalid_argument("invalid range");1151            }1152        }1153    ));1154    add_opt(common_arg(1155        {"--cpu-strict"}, "<0|1>",1156        string_format("use strict CPU placement (default: %u)\n", (unsigned) params.cpuparams.strict_cpu),1157        [](common_params & params, const std::string & value) {1158            params.cpuparams.strict_cpu = std::stoul(value);1159        }1160    ));1161    add_opt(common_arg(1162        {"--prio"}, "N",1163        string_format("set process/thread priority : low(-1), normal(0), medium(1), high(2), realtime(3) (default: %d)\n", params.cpuparams.priority),1164        [](common_params & params, int prio) {1165            if (prio < GGML_SCHED_PRIO_LOW || prio > GGML_SCHED_PRIO_REALTIME) {1166                throw std::invalid_argument("invalid value");1167            }1168            params.cpuparams.priority = (enum ggml_sched_priority) prio;1169        }1170    ));1171    add_opt(common_arg(1172        {"--poll"}, "<0...100>",1173        string_format("use polling level to wait for work (0 - no polling, default: %u)\n", (unsigned) params.cpuparams.poll),1174        [](common_params & params, const std::string & value) {1175            params.cpuparams.poll = std::stoul(value);1176        }1177    ));1178    add_opt(common_arg(1179        {"-Cb", "--cpu-mask-batch"}, "M",1180        "CPU affinity mask: arbitrarily long hex. Complements cpu-range-batch (default: same as --cpu-mask)",1181        [](common_params & params, const std::string & mask) {1182            params.cpuparams_batch.mask_valid = true;1183            if (!parse_cpu_mask(mask, params.cpuparams_batch.cpumask)) {1184                throw std::invalid_argument("invalid cpumask");1185            }1186        }1187    ));1188    add_opt(common_arg(1189        {"-Crb", "--cpu-range-batch"}, "lo-hi",1190        "ranges of CPUs for affinity. Complements --cpu-mask-batch",1191        [](common_params & params, const std::string & range) {1192            params.cpuparams_batch.mask_valid = true;1193            if (!parse_cpu_range(range, params.cpuparams_batch.cpumask)) {1194                throw std::invalid_argument("invalid range");1195            }1196        }1197    ));1198    add_opt(common_arg(1199        {"--cpu-strict-batch"}, "<0|1>",1200        "use strict CPU placement (default: same as --cpu-strict)",

Showing the first 1,200 of 3928 lines. Download the file for the rest.