echodict/llama.cpp
version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786
0479
1#include "arg.h"2 3#include "build-info.h"4#include "chat.h"5#include "common.h"6#include "download.h"7#include "hf-cache.h"8#include "json-schema-to-grammar.h"9#include "log.h"10#include "sampling.h"11#include "speculative.h"12#include "preset.h"13 14// fix problem with std::min and std::max15#if defined(_WIN32)16#define WIN32_LEAN_AND_MEAN17#ifndef NOMINMAX18# define NOMINMAX19#endif20#include <windows.h>21#endif22 23#define JSON_ASSERT GGML_ASSERT24#include <nlohmann/json.hpp>25 26#include <algorithm>27#include <cinttypes>28#include <climits>29#include <cstdarg>30#include <fstream>31#include <list>32#include <regex>33#include <set>34#include <string>35#include <thread> // for hardware_concurrency36#include <vector>37 38#ifndef __EMSCRIPTEN__39#ifdef __linux__40#include <linux/limits.h>41#elif defined(_WIN32)42# if !defined(PATH_MAX)43# define PATH_MAX MAX_PATH44# endif45#elif defined(_AIX)46#include <sys/limits.h>47#else48#include <sys/syslimits.h>49#endif50#endif51 52#define LLAMA_MAX_URL_LENGTH 2084 // Maximum URL Length in Chrome: 208353 54extern const char * LICENSES[];55 56using json = nlohmann::ordered_json;57using namespace common_arg_utils;58 59static std::initializer_list<enum llama_example> mmproj_examples = {60 LLAMA_EXAMPLE_MTMD,61 LLAMA_EXAMPLE_SERVER,62 LLAMA_EXAMPLE_CLI,63};64 65static std::string read_file(const std::string & fname) {66 std::ifstream file(fname);67 if (!file) {68 throw std::runtime_error(string_format("error: failed to open file '%s'\n", fname.c_str()));69 }70 std::string content((std::istreambuf_iterator<char>(file)), std::istreambuf_iterator<char>());71 file.close();72 return content;73}74 75static const std::vector<common_arg> & get_common_arg_defs() {76 static const std::vector<common_arg> options = [] {77 common_params params;78 auto ctx = common_params_parser_init(params, LLAMA_EXAMPLE_SERVER, nullptr);79 return ctx.options;80 }();81 return options;82}83 84common_arg & common_arg::set_examples(std::initializer_list<enum llama_example> examples) {85 this->examples = examples;86 return *this;87}88 89common_arg & common_arg::set_excludes(std::initializer_list<enum llama_example> excludes) {90 this->excludes = excludes;91 return *this;92}93 94common_arg & common_arg::set_env(const char * env) {95 help = help + "\n(env: " + env + ")";96 this->env = env;97 return *this;98}99 100common_arg & common_arg::set_sparam() {101 is_sparam = true;102 return *this;103}104 105common_arg & common_arg::set_preset_only() {106 is_preset_only = true;107 return *this;108}109 110bool common_arg::in_example(enum llama_example ex) {111 return examples.find(ex) != examples.end();112}113 114bool common_arg::is_exclude(enum llama_example ex) {115 return excludes.find(ex) != excludes.end();116}117 118bool common_arg::get_value_from_env(std::string & output) const {119 if (env == nullptr) return false;120 if (!args_neg.empty()) {121 // for compatibility, we need to check LLAMA_ARG_NO_ env as well122 std::string neg_env = env;123 string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");124 char * neg_value = std::getenv(neg_env.c_str());125 if (neg_value) {126 output = "0"; // falsey127 return true;128 }129 }130 char * value = std::getenv(env);131 if (value) {132 output = value;133 return true;134 }135 return false;136}137 138bool common_arg::has_value_from_env() const {139 if (env != nullptr && !args_neg.empty()) {140 // for compatibility, we need to check LLAMA_ARG_NO_ env as well141 std::string neg_env = env;142 string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");143 if (std::getenv(neg_env.c_str())) {144 return true;145 }146 }147 return env != nullptr && std::getenv(env);148}149 150static std::vector<std::string> break_str_into_lines(std::string input, size_t max_char_per_line) {151 std::vector<std::string> result;152 std::istringstream iss(input);153 std::string line;154 auto add_line = [&](const std::string& l) {155 if (l.length() <= max_char_per_line) {156 result.push_back(l);157 } else {158 std::istringstream line_stream(l);159 std::string word, current_line;160 while (line_stream >> word) {161 if (current_line.length() + !current_line.empty() + word.length() > max_char_per_line) {162 if (!current_line.empty()) result.push_back(current_line);163 current_line = word;164 } else {165 current_line += (!current_line.empty() ? " " : "") + word;166 }167 }168 if (!current_line.empty()) result.push_back(current_line);169 }170 };171 while (std::getline(iss, line)) {172 add_line(line);173 }174 return result;175}176 177std::string common_arg::to_string() const {178 // params for printing to console179 const static int n_leading_spaces = 40;180 const static int n_char_per_line_help = 70; // TODO: detect this based on current console181 std::string leading_spaces(n_leading_spaces, ' ');182 183 std::ostringstream ss;184 auto all_args = get_args(); // also contains args_neg185 for (const auto & arg : all_args) {186 if (arg == all_args.front()) {187 if (all_args.size() == 1) {188 ss << arg;189 } else {190 // first arg is usually abbreviation, we need padding to make it more beautiful191 auto tmp = std::string(arg) + ", ";192 auto spaces = std::string(std::max(0, 7 - (int)tmp.size()), ' ');193 ss << tmp << spaces;194 }195 } else {196 ss << arg << (arg != all_args.back() ? ", " : "");197 }198 }199 if (value_hint) ss << " " << value_hint;200 if (value_hint_2) ss << " " << value_hint_2;201 if (ss.tellp() > n_leading_spaces - 3) {202 // current line is too long, add new line203 ss << "\n" << leading_spaces;204 } else {205 // padding between arg and help, same line206 ss << std::string(leading_spaces.size() - ss.tellp(), ' ');207 }208 const auto help_lines = break_str_into_lines(help, n_char_per_line_help);209 for (const auto & line : help_lines) {210 ss << (&line == &help_lines.front() ? "" : leading_spaces) << line << "\n";211 }212 return ss.str();213}214 215std::vector<std::string> common_arg::get_args() const {216 std::vector<std::string> result;217 for (const auto & arg : args) {218 result.push_back(std::string(arg));219 }220 for (const auto & arg : args_neg) {221 result.push_back(std::string(arg));222 }223 return result;224}225 226std::vector<std::string> common_arg::get_env() const {227 std::vector<std::string> result;228 if (env) {229 result.push_back(std::string(env));230 }231 if (!args_neg.empty() && env) {232 // for compatibility, we need to add LLAMA_ARG_NO_ variant233 std::string neg_env = env;234 string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");235 result.push_back(neg_env);236 }237 return result;238}239 240//241// utils242//243 244// Helper function to parse tensor buffer override strings245static void parse_tensor_buffer_overrides(const std::string & value, std::vector<llama_model_tensor_buft_override> & overrides) {246 std::map<std::string, ggml_backend_buffer_type_t> buft_list;247 for (size_t i = 0; i < ggml_backend_dev_count(); ++i) {248 auto * dev = ggml_backend_dev_get(i);249 auto * buft = ggml_backend_dev_buffer_type(dev);250 if (buft) {251 buft_list[ggml_backend_buft_name(buft)] = buft;252 }253 }254 255 for (const auto & override : string_split<std::string>(value, ',')) {256 std::string::size_type pos = override.find('=');257 if (pos == std::string::npos) {258 throw std::invalid_argument("invalid value");259 }260 std::string tensor_name = override.substr(0, pos);261 std::string buffer_type = override.substr(pos + 1);262 263 if (buft_list.find(buffer_type) == buft_list.end()) {264 printf("Available buffer types:\n");265 for (const auto & it : buft_list) {266 printf(" %s\n", ggml_backend_buft_name(it.second));267 }268 throw std::invalid_argument("unknown buffer type");269 }270 // keep strings alive and avoid leaking memory by storing them in a static vector271 static std::list<std::string> buft_overrides;272 buft_overrides.push_back(tensor_name);273 overrides.push_back({buft_overrides.back().c_str(), buft_list.at(buffer_type)});274 }275}276 277static std::string clean_file_name(const std::string & fname) {278 std::string clean_fname = fname;279 string_replace_all(clean_fname, "\\", "_");280 string_replace_all(clean_fname, "/", "_");281 return clean_fname;282}283 284static bool common_params_handle_remote_preset(common_params & params, llama_example ex) {285 GGML_ASSERT(!params.model.hf_repo.empty());286 287 // the returned hf_repo is without tag288 auto [hf_repo, hf_tag] = common_download_split_repo_tag(params.model.hf_repo);289 290 // "latest" tag (default if not specified) is translated to "default" preset291 if (hf_tag == "latest") {292 hf_tag = "default";293 }294 295 std::string model_endpoint = common_get_model_endpoint();296 auto preset_url = model_endpoint + hf_repo + "/resolve/main/preset.ini";297 298 // prepare local path for caching299 auto preset_fname = clean_file_name(hf_repo + "_preset.ini");300 auto preset_path = fs_get_cache_file(preset_fname);301 common_download_opts opts;302 opts.bearer_token = params.hf_token;303 opts.offline = params.offline;304 const int status = common_download_file_single(preset_url, preset_path, opts);305 const bool has_preset = status >= 200 && status < 400;306 307 // remote preset is optional, so we don't error out if not found308 if (has_preset) {309 LOG_INF("applying remote preset from %s\n", preset_url.c_str());310 common_preset_context ctx(ex, /* only_remote_allowed */ true);311 common_preset global;312 auto remote_presets = ctx.load_from_ini(preset_path, global);313 remote_presets = ctx.cascade(global, remote_presets);314 if (remote_presets.find(hf_tag) != remote_presets.end()) {315 common_preset preset = remote_presets.at(hf_tag);316 LOG_INF("\n%s", preset.to_ini().c_str()); // to_ini already added trailing newline317 preset.apply_to_params(params);318 } else {319 throw std::runtime_error("Remote preset.ini does not contain [" + std::string(hf_tag) + "] section");320 }321 } else {322 LOG_INF("%s", "no remote preset found, skipping\n");323 }324 325 return has_preset;326}327 328struct handle_model_result {329 bool found_mmproj = false;330 common_params_model mmproj;331};332 333static handle_model_result common_params_handle_model(struct common_params_model & model,334 const std::string & bearer_token,335 bool offline) {336 handle_model_result result;337 338 if (!model.docker_repo.empty()) {339 model.path = common_docker_resolve_model(model.docker_repo);340 model.name = model.docker_repo;341 } else if (!model.hf_repo.empty()) {342 // If -m was used with -hf, treat the model "path" as the hf_file to download343 if (model.hf_file.empty() && !model.path.empty()) {344 model.hf_file = model.path;345 model.path = "";346 }347 common_download_opts opts;348 opts.bearer_token = bearer_token;349 opts.offline = offline;350 auto download_result = common_download_model(model, opts, true);351 352 if (download_result.model_path.empty()) {353 LOG_ERR("error: failed to download model from Hugging Face\n");354 exit(1);355 }356 357 model.name = model.hf_repo;358 model.path = download_result.model_path;359 360 if (!download_result.mmproj_path.empty()) {361 result.found_mmproj = true;362 result.mmproj.path = download_result.mmproj_path;363 }364 } else if (!model.url.empty()) {365 if (model.path.empty()) {366 auto f = string_split<std::string>(model.url, '#').front();367 f = string_split<std::string>(f, '?').front();368 model.path = fs_get_cache_file(string_split<std::string>(f, '/').back());369 }370 371 common_download_opts opts;372 opts.bearer_token = bearer_token;373 opts.offline = offline;374 auto download_result = common_download_model(model, opts);375 if (download_result.model_path.empty()) {376 LOG_ERR("error: failed to download model from %s\n", model.url.c_str());377 exit(1);378 }379 }380 381 return result;382}383 384const std::vector<ggml_type> kv_cache_types = {385 GGML_TYPE_F32,386 GGML_TYPE_F16,387 GGML_TYPE_BF16,388 GGML_TYPE_Q8_0,389 GGML_TYPE_Q4_0,390 GGML_TYPE_Q4_1,391 GGML_TYPE_IQ4_NL,392 GGML_TYPE_Q5_0,393 GGML_TYPE_Q5_1,394};395 396static ggml_type kv_cache_type_from_str(const std::string & s) {397 for (const auto & type : kv_cache_types) {398 if (ggml_type_name(type) == s) {399 return type;400 }401 }402 throw std::runtime_error("Unsupported cache type: " + s);403}404 405static std::string get_all_kv_cache_types() {406 std::ostringstream msg;407 for (const auto & type : kv_cache_types) {408 msg << ggml_type_name(type) << (&type == &kv_cache_types.back() ? "" : ", ");409 }410 return msg.str();411}412 413static bool parse_bool_value(const std::string & value) {414 if (is_truthy(value)) {415 return true;416 } else if (is_falsey(value)) {417 return false;418 } else {419 throw std::invalid_argument("invalid boolean value");420 }421}422 423//424// CLI argument parsing functions425//426 427static bool common_params_parse_ex(int argc, char ** argv, common_params_context & ctx_arg) {428 common_params & params = ctx_arg.params;429 430 // setup log directly from params.verbosity: see tools/cli/cli.cpp431 common_log_set_verbosity_thold(params.verbosity);432 433 std::unordered_map<std::string, std::pair<common_arg *, bool>> arg_to_options;434 for (auto & opt : ctx_arg.options) {435 for (const auto & arg : opt.args) {436 arg_to_options[arg] = {&opt, /* is_positive */ true};437 }438 for (const auto & arg : opt.args_neg) {439 arg_to_options[arg] = {&opt, /* is_positive */ false};440 }441 }442 443 // handle environment variables444 for (auto & opt : ctx_arg.options) {445 std::string value;446 if (opt.get_value_from_env(value)) {447 try {448 if (opt.handler_void && is_truthy(value)) {449 opt.handler_void(params);450 }451 if (opt.handler_int) {452 opt.handler_int(params, std::stoi(value));453 }454 if (opt.handler_bool) {455 opt.handler_bool(params, parse_bool_value(value));456 }457 if (opt.handler_string) {458 opt.handler_string(params, value);459 continue;460 }461 } catch (std::exception & e) {462 throw std::invalid_argument(string_format(463 "error while handling environment variable \"%s\": %s\n\n", opt.env, e.what()));464 }465 }466 }467 468 // handle command line arguments469 auto check_arg = [&](int i) {470 if (i+1 >= argc) {471 throw std::invalid_argument("expected value for argument");472 }473 };474 475 auto parse_cli_args = [&]() {476 std::set<std::string> seen_args;477 478 for (int i = 1; i < argc; i++) {479 const std::string arg_prefix = "--";480 481 std::string arg = argv[i];482 if (arg.compare(0, arg_prefix.size(), arg_prefix) == 0) {483 std::replace(arg.begin(), arg.end(), '_', '-');484 }485 if (arg_to_options.find(arg) == arg_to_options.end()) {486 throw std::invalid_argument(string_format("error: invalid argument: %s", arg.c_str()));487 }488 if (!seen_args.insert(arg).second) {489 LOG_WRN("DEPRECATED: argument '%s' specified multiple times, use comma-separated values instead (only last value will be used)\n", arg.c_str());490 }491 auto & tmp = arg_to_options[arg];492 auto opt = *tmp.first;493 bool is_positive = tmp.second;494 if (opt.has_value_from_env()) {495 fprintf(stderr, "warn: %s environment variable is set, but will be overwritten by command line argument %s\n", opt.env, arg.c_str());496 }497 try {498 if (opt.handler_void) {499 opt.handler_void(params);500 continue;501 }502 if (opt.handler_bool) {503 opt.handler_bool(params, is_positive);504 continue;505 }506 507 // arg with single value508 check_arg(i);509 std::string val = argv[++i];510 if (opt.handler_int) {511 opt.handler_int(params, std::stoi(val));512 continue;513 }514 if (opt.handler_string) {515 opt.handler_string(params, val);516 continue;517 }518 519 // arg with 2 values520 check_arg(i);521 std::string val2 = argv[++i];522 if (opt.handler_str_str) {523 opt.handler_str_str(params, val, val2);524 continue;525 }526 } catch (std::exception & e) {527 throw std::invalid_argument(string_format(528 "error while handling argument \"%s\": %s\n\n"529 "usage:\n%s\n\nto show complete usage, run with -h",530 arg.c_str(), e.what(), opt.to_string().c_str()));531 }532 }533 };534 535 // parse the first time to get -hf option (used for remote preset)536 parse_cli_args();537 538 // TODO: Remove later539 try {540 hf_cache::migrate_old_cache_to_hf_cache(params.hf_token, params.offline);541 } catch (const std::exception & e) {542 LOG_WRN("HF cache migration failed: %s\n", e.what());543 }544 // export_graph_ops loads only metadata545 const bool skip_model_download = ctx_arg.ex == LLAMA_EXAMPLE_EXPORT_GRAPH_OPS;546 547 // maybe handle remote preset548 if (!params.model.hf_repo.empty() && !skip_model_download) {549 std::string cli_hf_repo = params.model.hf_repo;550 bool has_preset = common_params_handle_remote_preset(params, ctx_arg.ex);551 552 // special case: if hf_repo explicitly set by preset, we need to preserve it (ignore CLI value)553 // this is useful when we have one HF repo pointing to other HF repos (one model - multiple GGUFs)554 std::string preset_hf_repo = params.model.hf_repo;555 bool preset_has_hf_repo = preset_hf_repo != cli_hf_repo;556 557 if (has_preset) {558 // re-parse CLI args to override preset values559 parse_cli_args();560 }561 562 // preserve hf_repo from preset if needed563 if (preset_has_hf_repo) {564 params.model.hf_repo = preset_hf_repo;565 }566 }567 568 postprocess_cpu_params(params.cpuparams, nullptr);569 postprocess_cpu_params(params.cpuparams_batch, ¶ms.cpuparams);570 571 postprocess_cpu_params(params.speculative.cpuparams, ¶ms.cpuparams);572 postprocess_cpu_params(params.speculative.cpuparams_batch, ¶ms.cpuparams_batch);573 574 if (params.prompt_cache_all && (params.interactive || params.interactive_first)) {575 throw std::invalid_argument("error: --prompt-cache-all not supported in interactive mode yet\n");576 }577 578 // handle model and download579 if (!skip_model_download) {580 auto res = common_params_handle_model(params.model, params.hf_token, params.offline);581 if (params.no_mmproj) {582 params.mmproj = {};583 } else if (res.found_mmproj && params.mmproj.path.empty() && params.mmproj.url.empty()) {584 // optionally, handle mmproj model when -hf is specified585 params.mmproj = res.mmproj;586 }587 // only download mmproj if the current example is using it588 for (const auto & ex : mmproj_examples) {589 if (ctx_arg.ex == ex) {590 common_params_handle_model(params.mmproj, params.hf_token, params.offline);591 break;592 }593 }594 common_params_handle_model(params.speculative.mparams_dft, params.hf_token, params.offline);595 common_params_handle_model(params.vocoder.model, params.hf_token, params.offline);596 }597 598 // model is required (except for server)599 // TODO @ngxson : maybe show a list of available models in CLI in this case600 if (params.model.path.empty() && ctx_arg.ex != LLAMA_EXAMPLE_SERVER && !skip_model_download && !params.usage && !params.completion) {601 throw std::invalid_argument("error: --model is required\n");602 }603 604 if (params.escape) {605 string_process_escapes(params.prompt);606 string_process_escapes(params.input_prefix);607 string_process_escapes(params.input_suffix);608 for (auto & antiprompt : params.antiprompt) {609 string_process_escapes(antiprompt);610 }611 for (auto & seq_breaker : params.sampling.dry_sequence_breakers) {612 string_process_escapes(seq_breaker);613 }614 for (auto & pair : params.speculative.replacements) {615 string_process_escapes(pair.first);616 string_process_escapes(pair.second);617 }618 }619 620 if (!params.kv_overrides.empty()) {621 params.kv_overrides.emplace_back();622 params.kv_overrides.back().key[0] = 0;623 }624 625 // pad tensor_buft_overrides for llama_params_fit:626 const size_t ntbo = llama_max_tensor_buft_overrides();627 while (params.tensor_buft_overrides.size() < ntbo) {628 params.tensor_buft_overrides.push_back({nullptr, nullptr});629 }630 631 if (!params.speculative.tensor_buft_overrides.empty()) {632 params.speculative.tensor_buft_overrides.push_back({nullptr, nullptr});633 }634 635 if (!params.chat_template.empty() && !common_chat_verify_template(params.chat_template, params.use_jinja)) {636 throw std::runtime_error(string_format(637 "error: the supplied chat template is not supported: %s%s\n",638 params.chat_template.c_str(),639 params.use_jinja ? "" : "\nnote: llama.cpp was started without --jinja, we only support commonly used templates"640 ));641 }642 643 return true;644}645 646static void common_params_print_usage(common_params_context & ctx_arg) {647 auto print_options = [](std::vector<common_arg *> & options) {648 for (common_arg * opt : options) {649 printf("%s", opt->to_string().c_str());650 }651 };652 653 std::vector<common_arg *> common_options;654 std::vector<common_arg *> sparam_options;655 std::vector<common_arg *> specific_options;656 for (auto & opt : ctx_arg.options) {657 // in case multiple LLAMA_EXAMPLE_* are set, we prioritize the LLAMA_EXAMPLE_* matching current example658 if (opt.is_sparam) {659 sparam_options.push_back(&opt);660 } else if (opt.in_example(ctx_arg.ex)) {661 specific_options.push_back(&opt);662 } else {663 common_options.push_back(&opt);664 }665 }666 printf("----- common params -----\n\n");667 print_options(common_options);668 printf("\n\n----- sampling params -----\n\n");669 print_options(sparam_options);670 // TODO: maybe convert enum llama_example to string671 printf("\n\n----- example-specific params -----\n\n");672 print_options(specific_options);673}674 675static void common_params_print_completion(common_params_context & ctx_arg) {676 std::vector<common_arg *> common_options;677 std::vector<common_arg *> sparam_options;678 std::vector<common_arg *> specific_options;679 680 for (auto & opt : ctx_arg.options) {681 if (opt.is_sparam) {682 sparam_options.push_back(&opt);683 } else if (opt.in_example(ctx_arg.ex)) {684 specific_options.push_back(&opt);685 } else {686 common_options.push_back(&opt);687 }688 }689 690 printf("_llama_completions() {\n");691 printf(" local cur prev opts\n");692 printf(" COMPREPLY=()\n");693 printf(" cur=\"${COMP_WORDS[COMP_CWORD]}\"\n");694 printf(" prev=\"${COMP_WORDS[COMP_CWORD-1]}\"\n\n");695 696 printf(" opts=\"");697 auto print_options = [](const std::vector<common_arg *> & options) {698 for (const common_arg * opt : options) {699 for (const char * arg : opt->args) {700 printf("%s ", arg);701 }702 }703 };704 705 print_options(common_options);706 print_options(sparam_options);707 print_options(specific_options);708 printf("\"\n\n");709 710 printf(" case \"$prev\" in\n");711 printf(" --model|-m)\n");712 printf(" COMPREPLY=( $(compgen -f -X '!*.gguf' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");713 printf(" return 0\n");714 printf(" ;;\n");715 printf(" --grammar-file)\n");716 printf(" COMPREPLY=( $(compgen -f -X '!*.gbnf' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");717 printf(" return 0\n");718 printf(" ;;\n");719 printf(" --chat-template-file)\n");720 printf(" COMPREPLY=( $(compgen -f -X '!*.jinja' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");721 printf(" return 0\n");722 printf(" ;;\n");723 printf(" *)\n");724 printf(" COMPREPLY=( $(compgen -W \"${opts}\" -- \"$cur\") )\n");725 printf(" return 0\n");726 printf(" ;;\n");727 printf(" esac\n");728 printf("}\n\n");729 730 std::set<std::string> executables = {731 "llama-batched",732 "llama-batched-bench",733 "llama-bench",734 "llama-cli",735 "llama-completion",736 "llama-convert-llama2c-to-ggml",737 "llama-cvector-generator",738 "llama-debug",739 "llama-diffusion-cli",740 "llama-embedding",741 "llama-eval-callback",742 "llama-export-lora",743 "llama-finetune",744 "llama-fit-params",745 "llama-gemma3-cli",746 "llama-gen-docs",747 "llama-gguf",748 "llama-gguf-hash",749 "llama-gguf-split",750 "llama-idle",751 "llama-imatrix",752 "llama-llava-cli",753 "llama-lookahead",754 "llama-lookup",755 "llama-lookup-create",756 "llama-lookup-merge",757 "llama-lookup-stats",758 "llama-minicpmv-cli",759 "llama-mtmd-cli",760 "llama-parallel",761 "llama-passkey",762 "llama-perplexity",763 "llama-q8dot",764 "llama-quantize",765 "llama-qwen2vl-cli",766 "llama-retrieval",767 "llama-save-load-state",768 "llama-server",769 "llama-simple",770 "llama-simple-chat",771 "llama-speculative",772 "llama-speculative-simple",773 "llama-tokenize",774 "llama-tts",775 "llama-vdot"776 };777 778 for (const auto& exe : executables) {779 printf("complete -F _llama_completions %s\n", exe.c_str());780 }781}782 783static std::vector<ggml_backend_dev_t> parse_device_list(const std::string & value) {784 std::vector<ggml_backend_dev_t> devices;785 auto dev_names = string_split<std::string>(value, ',');786 if (dev_names.empty()) {787 throw std::invalid_argument("no devices specified");788 }789 if (dev_names.size() == 1 && dev_names[0] == "none") {790 devices.push_back(nullptr);791 } else {792 for (const auto & device : dev_names) {793 auto * dev = ggml_backend_dev_by_name(device.c_str());794 if (!dev || ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_CPU) {795 throw std::invalid_argument(string_format("invalid device: %s", device.c_str()));796 }797 devices.push_back(dev);798 }799 devices.push_back(nullptr);800 }801 return devices;802}803 804static void add_rpc_devices(const std::string & servers) {805 auto rpc_servers = string_split<std::string>(servers, ',');806 if (rpc_servers.empty()) {807 throw std::invalid_argument("no RPC servers specified");808 }809 ggml_backend_reg_t rpc_reg = ggml_backend_reg_by_name("RPC");810 if (!rpc_reg) {811 throw std::invalid_argument("failed to find RPC backend");812 }813 typedef ggml_backend_reg_t (*ggml_backend_rpc_add_server_t)(const char * endpoint);814 ggml_backend_rpc_add_server_t ggml_backend_rpc_add_server_fn = (ggml_backend_rpc_add_server_t) ggml_backend_reg_get_proc_address(rpc_reg, "ggml_backend_rpc_add_server");815 if (!ggml_backend_rpc_add_server_fn) {816 throw std::invalid_argument("failed to find RPC add server function");817 }818 for (const auto & server : rpc_servers) {819 auto reg = ggml_backend_rpc_add_server_fn(server.c_str());820 ggml_backend_register(reg);821 }822}823 824bool common_params_to_map(int argc, char ** argv, llama_example ex, std::map<common_arg, std::string> & out_map) {825 common_params dummy_params;826 common_params_context ctx_arg = common_params_parser_init(dummy_params, ex, nullptr);827 828 std::unordered_map<std::string, common_arg *> arg_to_options;829 for (auto & opt : ctx_arg.options) {830 for (const auto & arg : opt.args) {831 arg_to_options[arg] = &opt;832 }833 for (const auto & arg : opt.args_neg) {834 arg_to_options[arg] = &opt;835 }836 }837 838 // TODO @ngxson : find a way to deduplicate this code839 840 // handle command line arguments841 auto check_arg = [&](int i) {842 if (i+1 >= argc) {843 throw std::invalid_argument("expected value for argument");844 }845 };846 847 std::set<std::string> seen_args;848 849 for (int i = 1; i < argc; i++) {850 const std::string arg_prefix = "--";851 852 std::string arg = argv[i];853 if (arg.compare(0, arg_prefix.size(), arg_prefix) == 0) {854 std::replace(arg.begin(), arg.end(), '_', '-');855 }856 if (arg_to_options.find(arg) == arg_to_options.end()) {857 throw std::invalid_argument(string_format("error: invalid argument: %s", arg.c_str()));858 }859 if (!seen_args.insert(arg).second) {860 LOG_WRN("DEPRECATED: argument '%s' specified multiple times, use comma-separated values instead (only last value will be used)\n", arg.c_str());861 }862 auto opt = *arg_to_options[arg];863 std::string val;864 if (opt.value_hint == nullptr && opt.value_hint_2 == nullptr) {865 // bool arg (need to reverse the meaning for negative args)866 bool is_neg = std::find(opt.args_neg.begin(), opt.args_neg.end(), arg) != opt.args_neg.end();867 val = is_neg ? "0" : "1";868 }869 if (opt.value_hint != nullptr) {870 // arg with single value871 check_arg(i);872 val = argv[++i];873 }874 if (opt.value_hint_2 != nullptr) {875 // TODO: support arg with 2 values876 throw std::invalid_argument("error: argument with 2 values is not yet supported\n");877 }878 out_map[opt] = val;879 }880 881 return true;882}883 884bool common_params_parse(int argc, char ** argv, common_params & params, llama_example ex, void(*print_usage)(int, char **)) {885 auto ctx_arg = common_params_parser_init(params, ex, print_usage);886 const common_params params_org = ctx_arg.params; // the example can modify the default params887 888 try {889 if (!common_params_parse_ex(argc, argv, ctx_arg)) {890 ctx_arg.params = params_org;891 return false;892 }893 if (ctx_arg.params.usage) {894 common_params_print_usage(ctx_arg);895 if (ctx_arg.print_usage) {896 ctx_arg.print_usage(argc, argv);897 }898 exit(0);899 }900 if (ctx_arg.params.completion) {901 common_params_print_completion(ctx_arg);902 exit(0);903 }904 params.lr.init();905 } catch (const std::invalid_argument & ex) {906 fprintf(stderr, "%s\n", ex.what());907 ctx_arg.params = params_org;908 return false;909 } catch (std::exception & ex) {910 fprintf(stderr, "%s\n", ex.what());911 exit(1); // for other exceptions, we exit with status code 1912 }913 914 return true;915}916 917static std::string list_builtin_chat_templates() {918 std::vector<const char *> supported_tmpl;919 int32_t res = llama_chat_builtin_templates(nullptr, 0);920 supported_tmpl.resize(res);921 res = llama_chat_builtin_templates(supported_tmpl.data(), supported_tmpl.size());922 std::ostringstream msg;923 for (auto & tmpl : supported_tmpl) {924 msg << tmpl << (&tmpl == &supported_tmpl.back() ? "" : ", ");925 }926 return msg.str();927}928 929bool common_arg_utils::is_truthy(const std::string & value) {930 return value == "on" || value == "enabled" || value == "true" || value == "1";931}932 933bool common_arg_utils::is_falsey(const std::string & value) {934 return value == "off" || value == "disabled" || value == "false" || value == "0";935}936 937bool common_arg_utils::is_autoy(const std::string & value) {938 return value == "auto" || value == "-1";939}940 941// Simple CSV parser that handles quoted fields and escaped quotes942// example:943// input: value1,"value, with, commas","value with ""escaped"" quotes",value4944// output: [value1] [value, with, commas] [value with "escaped" quotes] [value4]945static std::vector<std::string> parse_csv_row(const std::string& input) {946 std::vector<std::string> fields;947 std::string field;948 bool in_quotes = false;949 950 for (size_t i = 0; i < input.length(); ++i) {951 char ch = input[i];952 953 if (ch == '"') {954 if (!in_quotes) {955 // start of quoted field (only valid if at beginning of field)956 if (!field.empty()) {957 // quote appeared in middle of unquoted field, treat as literal958 field += '"';959 } else {960 in_quotes = true; // start961 }962 } else {963 if (i + 1 < input.length() && input[i + 1] == '"') {964 // escaped quote: ""965 field += '"';966 ++i; // skip the next quote967 } else {968 in_quotes = false; // end969 }970 }971 } else if (ch == ',') {972 if (in_quotes) {973 field += ',';974 } else {975 fields.push_back(std::move(field));976 field.clear();977 }978 } else {979 field += ch;980 }981 }982 983 // Add the last field984 fields.push_back(std::move(field));985 986 return fields;987}988 989common_params_context common_params_parser_init(common_params & params, llama_example ex, void(*print_usage)(int, char **)) {990 // per-example default params991 // we define here to make sure it's included in llama-gen-docs992 if (ex == LLAMA_EXAMPLE_COMPLETION) {993 params.use_jinja = false; // disable jinja by default994 995 } else if (ex == LLAMA_EXAMPLE_MTMD) {996 params.use_jinja = false; // disable jinja by default997 params.sampling.temp = 0.2; // lower temp by default for better quality998 999 } else if (ex == LLAMA_EXAMPLE_SERVER) {1000 params.n_parallel = -1; // auto by default1001 }1002 1003 params.use_color = tty_can_use_colors();1004 1005 // load dynamic backends1006 ggml_backend_load_all();1007 1008 common_params_context ctx_arg(params);1009 ctx_arg.print_usage = print_usage;1010 ctx_arg.ex = ex;1011 1012 std::string sampler_type_chars;1013 std::string sampler_type_names;1014 for (const auto & sampler : params.sampling.samplers) {1015 sampler_type_chars += common_sampler_type_to_chr(sampler);1016 sampler_type_names += common_sampler_type_to_str(sampler) + ";";1017 }1018 if (!sampler_type_names.empty()) {1019 sampler_type_names.pop_back(); // remove last semicolon1020 }1021 1022 1023 /**1024 * filter options by example1025 * rules:1026 * - all examples inherit options from LLAMA_EXAMPLE_COMMON1027 * - if LLAMA_EXAMPLE_* is set (other than COMMON), we only show the option in the corresponding example1028 * - if both {LLAMA_EXAMPLE_COMMON, LLAMA_EXAMPLE_*,} are set, we will prioritize the LLAMA_EXAMPLE_* matching current example1029 */1030 auto add_opt = [&](common_arg arg) {1031 if ((arg.in_example(ex) || arg.in_example(LLAMA_EXAMPLE_COMMON)) && !arg.is_exclude(ex)) {1032 ctx_arg.options.push_back(std::move(arg));1033 }1034 };1035 1036 1037 add_opt(common_arg(1038 {"-h", "--help", "--usage"},1039 "print usage and exit",1040 [](common_params & params) {1041 params.usage = true;1042 }1043 ));1044 add_opt(common_arg(1045 {"--version"},1046 "show version and build info",1047 [](common_params &) {1048 fprintf(stderr, "version: %d (%s)\n", llama_build_number(), llama_commit());1049 fprintf(stderr, "built with %s for %s\n", llama_compiler(), llama_build_target());1050 exit(0);1051 }1052 ));1053 add_opt(common_arg(1054 {"--license"},1055 "show source code license and dependencies",1056 [](common_params &) {1057 for (int i = 0; LICENSES[i]; ++i) {1058 printf("%s\n", LICENSES[i]);1059 }1060 exit(0);1061 }1062 ));1063 add_opt(common_arg(1064 {"-cl", "--cache-list"},1065 "show list of models in cache",1066 [](common_params &) {1067 auto models = common_list_cached_models();1068 printf("number of models in cache: %zu\n", models.size());1069 for (size_t i = 0; i < models.size(); i++) {1070 printf("%4zu. %s\n", i + 1, models[i].to_string().c_str());1071 }1072 exit(0);1073 }1074 ));1075 add_opt(common_arg(1076 {"--completion-bash"},1077 "print source-able bash completion script for llama.cpp",1078 [](common_params & params) {1079 params.completion = true;1080 }1081 ));1082 add_opt(common_arg(1083 {"--verbose-prompt"},1084 string_format("print a verbose prompt before generation (default: %s)", params.verbose_prompt ? "true" : "false"),1085 [](common_params & params) {1086 params.verbose_prompt = true;1087 }1088 ).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_EMBEDDING, LLAMA_EXAMPLE_RETRIEVAL}));1089 add_opt(common_arg(1090 {"--display-prompt"},1091 {"--no-display-prompt"},1092 string_format("whether to print prompt at generation (default: %s)", params.display_prompt ? "true" : "false"),1093 [](common_params & params, bool value) {1094 params.display_prompt = value;1095 }1096 ).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI}));1097 add_opt(common_arg(1098 {"-co", "--color"}, "[on|off|auto]",1099 "Colorize output to distinguish prompt and user input from generations ('on', 'off', or 'auto', default: 'auto')\n"1100 "'auto' enables colors when output is to a terminal",1101 [](common_params & params, const std::string & value) {1102 if (is_truthy(value)) {1103 params.use_color = true;1104 } else if (is_falsey(value)) {1105 params.use_color = false;1106 } else if (is_autoy(value)) {1107 params.use_color = tty_can_use_colors();1108 } else {1109 throw std::invalid_argument(1110 string_format("error: unknown value for --color: '%s'\n", value.c_str()));1111 }1112 }1113 ).set_examples({LLAMA_EXAMPLE_COMPLETION, LLAMA_EXAMPLE_CLI, LLAMA_EXAMPLE_SPECULATIVE, LLAMA_EXAMPLE_LOOKUP}));1114 add_opt(common_arg(1115 {"-t", "--threads"}, "N",1116 string_format("number of CPU threads to use during generation (default: %d)", params.cpuparams.n_threads),1117 [](common_params & params, int value) {1118 params.cpuparams.n_threads = value;1119 if (params.cpuparams.n_threads <= 0) {1120 params.cpuparams.n_threads = std::thread::hardware_concurrency();1121 }1122 }1123 ).set_env("LLAMA_ARG_THREADS"));1124 add_opt(common_arg(1125 {"-tb", "--threads-batch"}, "N",1126 "number of threads to use during batch and prompt processing (default: same as --threads)",1127 [](common_params & params, int value) {1128 params.cpuparams_batch.n_threads = value;1129 if (params.cpuparams_batch.n_threads <= 0) {1130 params.cpuparams_batch.n_threads = std::thread::hardware_concurrency();1131 }1132 }1133 ));1134 add_opt(common_arg(1135 {"-C", "--cpu-mask"}, "M",1136 "CPU affinity mask: arbitrarily long hex. Complements cpu-range (default: \"\")",1137 [](common_params & params, const std::string & mask) {1138 params.cpuparams.mask_valid = true;1139 if (!parse_cpu_mask(mask, params.cpuparams.cpumask)) {1140 throw std::invalid_argument("invalid cpumask");1141 }1142 }1143 ));1144 add_opt(common_arg(1145 {"-Cr", "--cpu-range"}, "lo-hi",1146 "range of CPUs for affinity. Complements --cpu-mask",1147 [](common_params & params, const std::string & range) {1148 params.cpuparams.mask_valid = true;1149 if (!parse_cpu_range(range, params.cpuparams.cpumask)) {1150 throw std::invalid_argument("invalid range");1151 }1152 }1153 ));1154 add_opt(common_arg(1155 {"--cpu-strict"}, "<0|1>",1156 string_format("use strict CPU placement (default: %u)\n", (unsigned) params.cpuparams.strict_cpu),1157 [](common_params & params, const std::string & value) {1158 params.cpuparams.strict_cpu = std::stoul(value);1159 }1160 ));1161 add_opt(common_arg(1162 {"--prio"}, "N",1163 string_format("set process/thread priority : low(-1), normal(0), medium(1), high(2), realtime(3) (default: %d)\n", params.cpuparams.priority),1164 [](common_params & params, int prio) {1165 if (prio < GGML_SCHED_PRIO_LOW || prio > GGML_SCHED_PRIO_REALTIME) {1166 throw std::invalid_argument("invalid value");1167 }1168 params.cpuparams.priority = (enum ggml_sched_priority) prio;1169 }1170 ));1171 add_opt(common_arg(1172 {"--poll"}, "<0...100>",1173 string_format("use polling level to wait for work (0 - no polling, default: %u)\n", (unsigned) params.cpuparams.poll),1174 [](common_params & params, const std::string & value) {1175 params.cpuparams.poll = std::stoul(value);1176 }1177 ));1178 add_opt(common_arg(1179 {"-Cb", "--cpu-mask-batch"}, "M",1180 "CPU affinity mask: arbitrarily long hex. Complements cpu-range-batch (default: same as --cpu-mask)",1181 [](common_params & params, const std::string & mask) {1182 params.cpuparams_batch.mask_valid = true;1183 if (!parse_cpu_mask(mask, params.cpuparams_batch.cpumask)) {1184 throw std::invalid_argument("invalid cpumask");1185 }1186 }1187 ));1188 add_opt(common_arg(1189 {"-Crb", "--cpu-range-batch"}, "lo-hi",1190 "ranges of CPUs for affinity. Complements --cpu-mask-batch",1191 [](common_params & params, const std::string & range) {1192 params.cpuparams_batch.mask_valid = true;1193 if (!parse_cpu_range(range, params.cpuparams_batch.cpumask)) {1194 throw std::invalid_argument("invalid range");1195 }1196 }1197 ));1198 add_opt(common_arg(1199 {"--cpu-strict-batch"}, "<0|1>",1200 "use strict CPU placement (default: same as --cpu-strict)",