Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#include "arg.h"2 3#include "build-info.h"4#include "chat.h"5#include "common.h"6#include "download.h"7#include "json-schema-to-grammar.h"8#include "llama.h"9#include "log.h"10#include "sampling.h"11#include "speculative.h"12#include "preset.h"13 14// fix problem with std::min and std::max15#if defined(_WIN32)16#define WIN32_LEAN_AND_MEAN17#ifndef NOMINMAX18# define NOMINMAX19#endif20#include <windows.h>21#include <shellapi.h>22#endif23 24#define JSON_ASSERT GGML_ASSERT25#include <nlohmann/json.hpp>26 27#include <algorithm>28#include <cinttypes>29#include <climits>30#include <cmath>31#include <cstdarg>32#include <filesystem>33#include <fstream>34#include <list>35#include <regex>36#include <set>37#include <string>38#include <thread> // for hardware_concurrency39#include <vector>40 41#ifndef __EMSCRIPTEN__42#ifdef __linux__43#include <linux/limits.h>44#elif defined(_WIN32)45# if !defined(PATH_MAX)46# define PATH_MAX MAX_PATH47# endif48#elif defined(_AIX)49#include <sys/limits.h>50#else51#include <sys/syslimits.h>52#endif53#endif54 55#define LLAMA_MAX_URL_LENGTH 2084 // Maximum URL Length in Chrome: 208356 57using json = nlohmann::ordered_json;58using namespace common_arg_utils;59 60static std::initializer_list<enum llama_example> mmproj_examples = {61 LLAMA_EXAMPLE_MTMD,62 LLAMA_EXAMPLE_SERVER,63 LLAMA_EXAMPLE_CLI,64 LLAMA_EXAMPLE_TTS,65};66 67static std::string read_file(const std::string & fname) {68 std::ifstream file(fname);69 if (!file) {70 throw std::runtime_error(string_format("error: failed to open file '%s'\n", fname.c_str()));71 }72 std::string content((std::istreambuf_iterator<char>(file)), std::istreambuf_iterator<char>());73 file.close();74 return content;75}76 77static const std::vector<common_arg> & get_common_arg_defs() {78 static const std::vector<common_arg> options = [] {79 common_params params;80 auto ctx = common_params_parser_init(params, LLAMA_EXAMPLE_SERVER, nullptr);81 return ctx.options;82 }();83 return options;84}85 86common_arg & common_arg::set_examples(std::initializer_list<enum llama_example> examples) {87 this->examples = examples;88 return *this;89}90 91common_arg & common_arg::set_excludes(std::initializer_list<enum llama_example> excludes) {92 this->excludes = excludes;93 return *this;94}95 96common_arg & common_arg::set_env(const char * env) {97 help = help + "\n(env: " + env + ")";98 this->env = env;99 return *this;100}101 102common_arg & common_arg::set_sampling() {103 is_sampling = true;104 return *this;105}106 107common_arg & common_arg::set_spec() {108 is_spec = true;109 return *this;110}111 112common_arg & common_arg::set_preset_only() {113 is_preset_only = true;114 return *this;115}116 117bool common_arg::in_example(enum llama_example ex) {118 return examples.find(ex) != examples.end();119}120 121bool common_arg::is_exclude(enum llama_example ex) {122 return excludes.find(ex) != excludes.end();123}124 125bool common_arg::get_value_from_env(std::string & output) const {126 if (env == nullptr) return false;127 if (!args_neg.empty()) {128 // for compatibility, we need to check LLAMA_ARG_NO_ env as well129 std::string neg_env = env;130 string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");131 char * neg_value = std::getenv(neg_env.c_str());132 if (neg_value) {133 output = "0"; // falsey134 return true;135 }136 }137 char * value = std::getenv(env);138 if (value) {139 output = value;140 return true;141 }142 return false;143}144 145bool common_arg::has_value_from_env() const {146 if (env != nullptr && !args_neg.empty()) {147 // for compatibility, we need to check LLAMA_ARG_NO_ env as well148 std::string neg_env = env;149 string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");150 if (std::getenv(neg_env.c_str())) {151 return true;152 }153 }154 return env != nullptr && std::getenv(env);155}156 157static std::vector<std::string> break_str_into_lines(std::string input, size_t max_char_per_line) {158 std::vector<std::string> result;159 std::istringstream iss(input);160 std::string line;161 auto add_line = [&](const std::string& l) {162 if (l.length() <= max_char_per_line) {163 result.push_back(l);164 } else {165 std::istringstream line_stream(l);166 std::string word, current_line;167 while (line_stream >> word) {168 if (current_line.length() + !current_line.empty() + word.length() > max_char_per_line) {169 if (!current_line.empty()) result.push_back(current_line);170 current_line = word;171 } else {172 current_line += (!current_line.empty() ? " " : "") + word;173 }174 }175 if (!current_line.empty()) result.push_back(current_line);176 }177 };178 while (std::getline(iss, line)) {179 add_line(line);180 }181 return result;182}183 184std::string common_arg::to_string() const {185 // params for printing to console186 const static int n_leading_spaces = 40;187 const static int n_char_per_line_help = 70; // TODO: detect this based on current console188 std::string leading_spaces(n_leading_spaces, ' ');189 190 std::ostringstream ss;191 auto all_args = get_args(); // also contains args_neg192 for (const auto & arg : all_args) {193 if (arg == all_args.front()) {194 if (all_args.size() == 1) {195 ss << arg;196 } else {197 // first arg is usually abbreviation, we need padding to make it more beautiful198 auto tmp = std::string(arg) + ", ";199 auto spaces = std::string(std::max(0, 7 - (int)tmp.size()), ' ');200 ss << tmp << spaces;201 }202 } else {203 ss << arg << (arg != all_args.back() ? ", " : "");204 }205 }206 if (value_hint) ss << " " << value_hint;207 if (value_hint_2) ss << " " << value_hint_2;208 if (ss.tellp() > n_leading_spaces - 3) {209 // current line is too long, add new line210 ss << "\n" << leading_spaces;211 } else {212 // padding between arg and help, same line213 ss << std::string(leading_spaces.size() - ss.tellp(), ' ');214 }215 const auto help_lines = break_str_into_lines(help, n_char_per_line_help);216 for (const auto & line : help_lines) {217 ss << (&line == &help_lines.front() ? "" : leading_spaces) << line << "\n";218 }219 return ss.str();220}221 222std::vector<std::string> common_arg::get_args() const {223 std::vector<std::string> result;224 for (const auto & arg : args) {225 result.push_back(std::string(arg));226 }227 for (const auto & arg : args_neg) {228 result.push_back(std::string(arg));229 }230 return result;231}232 233std::vector<std::string> common_arg::get_env() const {234 std::vector<std::string> result;235 if (env) {236 result.push_back(std::string(env));237 }238 if (!args_neg.empty() && env) {239 // for compatibility, we need to add LLAMA_ARG_NO_ variant240 std::string neg_env = env;241 string_replace_all(neg_env, "LLAMA_ARG_", "LLAMA_ARG_NO_");242 result.push_back(neg_env);243 }244 return result;245}246 247//248// utils249//250 251// Helper function to parse tensor buffer override strings252static void parse_tensor_buffer_overrides(const std::string & value, std::vector<llama_model_tensor_buft_override> & overrides) {253 ggml_backend_load_all();254 255 std::map<std::string, ggml_backend_buffer_type_t> buft_list;256 for (size_t i = 0; i < ggml_backend_dev_count(); ++i) {257 auto * dev = ggml_backend_dev_get(i);258 auto * buft = ggml_backend_dev_buffer_type(dev);259 if (buft) {260 buft_list[ggml_backend_buft_name(buft)] = buft;261 }262 }263 264 for (const auto & override : string_split<std::string>(value, ',')) {265 std::string::size_type pos = override.find('=');266 if (pos == std::string::npos) {267 throw std::invalid_argument("invalid value");268 }269 std::string tensor_name = override.substr(0, pos);270 std::string buffer_type = override.substr(pos + 1);271 272 if (buft_list.find(buffer_type) == buft_list.end()) {273 printf("Available buffer types:\n");274 for (const auto & it : buft_list) {275 printf(" %s\n", ggml_backend_buft_name(it.second));276 }277 throw std::invalid_argument("unknown buffer type");278 }279 // keep strings alive and avoid leaking memory by storing them in a static vector280 static std::list<std::string> buft_overrides;281 buft_overrides.push_back(tensor_name);282 overrides.push_back({buft_overrides.back().c_str(), buft_list.at(buffer_type)});283 }284}285 286static std::string clean_file_name(const std::string & fname) {287 std::string clean_fname = fname;288 string_replace_all(clean_fname, "\\", "_");289 string_replace_all(clean_fname, "/", "_");290 return clean_fname;291}292 293struct handle_model_result {294 bool found_mmproj = false;295 common_params_model mmproj;296 297 bool found_mtp = false;298 common_params_model mtp;299 300 bool found_preset = false;301 std::string preset_path;302};303 304const std::vector<ggml_type> kv_cache_types = {305 GGML_TYPE_F32,306 GGML_TYPE_F16,307 GGML_TYPE_BF16,308 GGML_TYPE_Q8_0,309 GGML_TYPE_Q4_0,310 GGML_TYPE_Q4_1,311 GGML_TYPE_IQ4_NL,312 GGML_TYPE_Q5_0,313 GGML_TYPE_Q5_1,314};315 316static ggml_type kv_cache_type_from_str(const std::string & s) {317 for (const auto & type : kv_cache_types) {318 if (ggml_type_name(type) == s) {319 return type;320 }321 }322 throw std::runtime_error("Unsupported cache type: " + s);323}324 325static std::string get_all_kv_cache_types() {326 std::ostringstream msg;327 for (const auto & type : kv_cache_types) {328 msg << ggml_type_name(type) << (&type == &kv_cache_types.back() ? "" : ", ");329 }330 return msg.str();331}332 333static bool parse_bool_value(const std::string & value) {334 if (is_truthy(value)) {335 return true;336 } else if (is_falsey(value)) {337 return false;338 } else {339 throw std::invalid_argument("invalid boolean value");340 }341}342 343[[noreturn]] static void arg_removed(const std::string & msg) {344 throw std::invalid_argument("the argument has been removed. " + msg);345}346 347//348// common_models_handler349//350 351static std::string get_default_local_path(const std::string & url) {352 auto f = string_split<std::string>(url, '#').front();353 f = string_split<std::string>(f, '?').front();354 return fs_get_cache_file(string_split<std::string>(f, '/').back());355}356 357static bool spec_types_is_default(const common_params & params) {358 return params.speculative.types == std::vector<enum common_speculative_type>{COMMON_SPECULATIVE_TYPE_NONE};359}360 361common_models_handler common_models_handler_init(const common_params & params, llama_example curr_ex) {362 common_download_hf_plan plan;363 common_download_hf_plan plan_spec;364 common_download_opts opts;365 366 const bool spec_type_draft_mtp = std::find(params.speculative.types.begin(),367 params.speculative.types.end(),368 COMMON_SPECULATIVE_TYPE_DRAFT_MTP) != params.speculative.types.end();369 370 const bool spec_type_draft_dflash = std::find(params.speculative.types.begin(),371 params.speculative.types.end(),372 COMMON_SPECULATIVE_TYPE_DRAFT_DFLASH) != params.speculative.types.end();373 374 const bool spec_type_draft_eagle3 = std::find(params.speculative.types.begin(),375 params.speculative.types.end(),376 COMMON_SPECULATIVE_TYPE_DRAFT_EAGLE3) != params.speculative.types.end();377 378 const bool spec_type_draft_dspark = std::find(params.speculative.types.begin(),379 params.speculative.types.end(),380 COMMON_SPECULATIVE_TYPE_DRAFT_DSPARK) != params.speculative.types.end();381 382 // only download mmproj if the current example is using it383 bool use_mmproj = false;384 for (const auto & ex : mmproj_examples) {385 if (curr_ex == ex) {386 use_mmproj = true;387 break;388 }389 }390 391 opts.bearer_token = params.hf_token;392 opts.offline = params.offline;393 opts.download_mtp = spec_type_draft_mtp;394 opts.download_eagle3 = spec_type_draft_eagle3;395 opts.download_dflash = spec_type_draft_dflash;396 opts.download_dspark = spec_type_draft_dspark;397 opts.download_mmproj = use_mmproj && !params.no_mmproj398 && params.mmproj.path.empty() && params.mmproj.url.empty();399 400 if (!params.model.hf_repo.empty()) {401 plan = common_download_get_hf_plan(params.model, opts);402 }403 404 if (!params.speculative.draft.mparams.hf_repo.empty()) {405 // without a requested type, discover every sidecar the draft repo ships to infer the type later406 auto opts_spec = opts;407 if (spec_types_is_default(params)) {408 opts_spec.download_mtp = true;409 opts_spec.download_dflash = true;410 opts_spec.download_eagle3 = true;411 opts_spec.download_dspark = true;412 }413 plan_spec = common_download_get_hf_plan(params.speculative.draft.mparams, opts_spec);414 }415 416 return common_models_handler{plan, plan_spec, opts};417}418 419bool common_models_handler_is_preset_repo(const common_models_handler & handler) {420 return !handler.plan.preset.url.empty();421}422 423static std::vector<common_download_task> build_url_tasks(const common_params_model & model, common_download_opts opts) {424 auto parts = common_download_get_all_parts(model.url);425 std::vector<common_download_task> tasks;426 427 // single-part: download straight to model.path if the user gave one (-m), else the cache default428 if (parts.size() == 1) {429 common_download_task task;430 task.url = parts[0];431 task.local_path = model.path.empty() ? get_default_local_path(parts[0]) : model.path;432 task.opts = opts;433 tasks.push_back(std::move(task));434 return tasks;435 }436 437 // multi-part: place each part under the user's -m directory (if given), else the cache default438 std::string base_dir;439 if (!model.path.empty()) {440 auto pos = model.path.rfind('/');441 base_dir = pos == std::string::npos ? std::string(".") : model.path.substr(0, pos);442 }443 444 for (const auto & part : parts) {445 common_download_task task;446 task.url = part;447 task.opts = opts;448 449 std::string local = get_default_local_path(part);450 if (!base_dir.empty()) {451 auto pos = local.rfind('/');452 std::string name = pos == std::string::npos ? local : local.substr(pos + 1);453 local = base_dir + "/" + name;454 }455 task.local_path = local;456 tasks.push_back(std::move(task));457 }458 return tasks;459}460 461void common_models_handler_apply(common_models_handler & handler, common_params & params, common_download_callback * callback) {462 std::vector<common_download_task> tasks;463 464 auto & plan = handler.plan;465 auto & plan_spec = handler.plan_spec;466 467 auto opts = handler.opts; // copy468 opts.callback = callback;469 470 // handle plain "url" if needed471 auto handle_url = [&](common_params_model & model) {472 if (!model.url.empty()) {473 if (model.path.empty()) {474 model.path = get_default_local_path(model.url);475 }476 }477 };478 handle_url(params.model);479 handle_url(params.mmproj);480 handle_url(params.speculative.draft.mparams);481 482 // optionally, if docker repo is set, resolve it483 if (!params.model.docker_repo.empty()) {484 params.model.url = common_docker_resolve_model(params.model.docker_repo);485 params.model.path = get_default_local_path(params.model.url);486 }487 488 // handle plain "url" tasks (non-hf)489 if (!params.model.url.empty()) {490 auto url_tasks = build_url_tasks(params.model, opts);491 // the first part is what gets loaded, so point params.model.path at it492 if (!url_tasks.empty()) {493 std::string first_path = url_tasks.front().local_path;494 url_tasks.front().on_done = [&, first_path]() { params.model.path = first_path; };495 }496 for (auto & task : url_tasks) {497 tasks.push_back(std::move(task));498 }499 }500 if (!params.mmproj.url.empty()) {501 common_download_task task;502 task.url = params.mmproj.url;503 task.local_path = params.mmproj.path;504 task.opts = opts;505 tasks.push_back(task);506 }507 bool had_spec_url = false;508 if (!params.speculative.draft.mparams.url.empty()) {509 common_download_task task;510 task.url = params.speculative.draft.mparams.url;511 task.local_path = params.speculative.draft.mparams.path;512 task.opts = opts;513 tasks.push_back(task);514 had_spec_url = true;515 }516 517 // handle hf_plan tasks518 auto add_tasks = [&opts, &tasks](const hf_cache::hf_files & model_files,519 const hf_cache::hf_file & primary,520 common_params_model & model) {521 for (size_t i = 0; i < model_files.size(); ++i) {522 auto & model_file = model_files[i];523 bool is_primary = (model_file.path == primary.path);524 tasks.emplace_back(model_file, opts, [&, is_primary]() {525 if (is_primary) {526 // the primary file is the first split (00001-of), use it as model path527 model.path = hf_cache::finalize_file(model_file);528 } else {529 hf_cache::finalize_file(model_file);530 }531 });532 }533 };534 535 // an explicit draft file selection (e.g. -md with -hfd) disables the sidecar resolution of the draft repo536 if (!params.speculative.draft.mparams.hf_file.empty()) {537 plan_spec.mtp = {};538 plan_spec.dflash = {};539 plan_spec.eagle3 = {};540 plan_spec.dspark = {};541 }542 543 // infer the speculative type from the sidecar shipped by the draft repo when none is requested544 if (spec_types_is_default(params)) {545 if (!plan_spec.mtp.local_path.empty()) {546 params.speculative.types = { COMMON_SPECULATIVE_TYPE_DRAFT_MTP };547 plan_spec.dspark = {};548 plan_spec.dflash = {};549 plan_spec.eagle3 = {};550 } else if (!plan_spec.dspark.local_path.empty()) {551 // dspark outranks dflash, its sidecar carries the extra Markov head552 params.speculative.types = { COMMON_SPECULATIVE_TYPE_DRAFT_DSPARK };553 plan_spec.dflash = {};554 plan_spec.eagle3 = {};555 } else if (!plan_spec.dflash.local_path.empty()) {556 params.speculative.types = { COMMON_SPECULATIVE_TYPE_DRAFT_DFLASH };557 plan_spec.eagle3 = {};558 } else if (!plan_spec.eagle3.local_path.empty()) {559 params.speculative.types = { COMMON_SPECULATIVE_TYPE_DRAFT_EAGLE3 };560 }561 }562 563 // when a sidecar type is requested, the draft repo resolves to its sidecar instead of a full model564 const bool spec_sidecar_found = !plan_spec.mtp.local_path.empty() ||565 !plan_spec.dflash.local_path.empty() ||566 !plan_spec.eagle3.local_path.empty() ||567 !plan_spec.dspark.local_path.empty();568 if (!plan_spec.mtp.local_path.empty() && !had_spec_url) {569 tasks.emplace_back(plan_spec.mtp, opts, [&]() {570 // only use the discovered MTP head when no draft path is set yet571 if (params.speculative.draft.mparams.path.empty()) {572 params.speculative.draft.mparams.path = hf_cache::finalize_file(plan_spec.mtp);573 } else {574 hf_cache::finalize_file(plan_spec.mtp);575 }576 });577 }578 if (!plan_spec.dflash.local_path.empty() && !had_spec_url) {579 tasks.emplace_back(plan_spec.dflash, opts, [&]() {580 // only use the discovered DFlash sidecar when no draft path is set yet581 if (params.speculative.draft.mparams.path.empty()) {582 params.speculative.draft.mparams.path = hf_cache::finalize_file(plan_spec.dflash);583 } else {584 hf_cache::finalize_file(plan_spec.dflash);585 }586 });587 }588 if (!plan_spec.eagle3.local_path.empty() && !had_spec_url) {589 tasks.emplace_back(plan_spec.eagle3, opts, [&]() {590 // only use the discovered Eagle3 sidecar when no draft path is set yet591 if (params.speculative.draft.mparams.path.empty()) {592 params.speculative.draft.mparams.path = hf_cache::finalize_file(plan_spec.eagle3);593 } else {594 hf_cache::finalize_file(plan_spec.eagle3);595 }596 });597 }598 if (!plan_spec.dspark.local_path.empty() && !had_spec_url) {599 tasks.emplace_back(plan_spec.dspark, opts, [&]() {600 // only use the discovered DSpark sidecar when no draft path is set yet601 if (params.speculative.draft.mparams.path.empty()) {602 params.speculative.draft.mparams.path = hf_cache::finalize_file(plan_spec.dspark);603 } else {604 hf_cache::finalize_file(plan_spec.dspark);605 }606 });607 }608 609 // a wired draft sidecar counts as an explicit draft for the main plan fallback below610 if (spec_sidecar_found) {611 had_spec_url = true;612 }613 614 // handle plan_spec (e.g. --spec-draft-hf)615 if (!plan_spec.model_files.empty() && !had_spec_url && !spec_sidecar_found) {616 add_tasks(plan_spec.model_files, plan_spec.primary, params.speculative.draft.mparams);617 had_spec_url = true;618 }619 620 if (!plan.model_files.empty()) {621 add_tasks(plan.model_files, plan.primary, params.model);622 }623 if (!plan.mmproj.local_path.empty()) {624 tasks.emplace_back(plan.mmproj, opts, [&]() {625 params.mmproj.path = hf_cache::finalize_file(plan.mmproj);626 });627 }628 if (!plan.mtp.local_path.empty() && !had_spec_url) {629 tasks.emplace_back(plan.mtp, opts, [&]() {630 // only fall back to the discovered MTP head when no draft was explicitly provided631 if (params.speculative.draft.mparams.empty()) {632 params.speculative.draft.mparams.path = hf_cache::finalize_file(plan.mtp);633 } else {634 hf_cache::finalize_file(plan.mtp);635 }636 });637 }638 if (!plan.dflash.local_path.empty() && !had_spec_url) {639 tasks.emplace_back(plan.dflash, opts, [&]() {640 // only fall back to the discovered DFlash sidecar when no draft was explicitly provided641 if (params.speculative.draft.mparams.empty()) {642 params.speculative.draft.mparams.path = hf_cache::finalize_file(plan.dflash);643 } else {644 hf_cache::finalize_file(plan.dflash);645 }646 });647 }648 if (!plan.eagle3.local_path.empty() && !had_spec_url) {649 tasks.emplace_back(plan.eagle3, opts, [&]() {650 // only fall back to the discovered Eagle3 sidecar when no draft was explicitly provided651 if (params.speculative.draft.mparams.empty()) {652 params.speculative.draft.mparams.path = hf_cache::finalize_file(plan.eagle3);653 } else {654 hf_cache::finalize_file(plan.eagle3);655 }656 });657 }658 if (!plan.dspark.local_path.empty() && !had_spec_url) {659 tasks.emplace_back(plan.dspark, opts, [&]() {660 // only fall back to the discovered DSpark sidecar when no draft was explicitly provided661 if (params.speculative.draft.mparams.empty()) {662 params.speculative.draft.mparams.path = hf_cache::finalize_file(plan.dspark);663 } else {664 hf_cache::finalize_file(plan.dspark);665 }666 });667 }668 if (!plan.preset.local_path.empty()) {669 tasks.emplace_back(plan.preset, opts, [&]() {670 // if HF repo is a preset repo, we simply run server in router mode with the preset.ini file671 params.models_preset_hf = params.model.hf_repo; // only for showing a warning672 params.models_preset = hf_cache::finalize_file(plan.preset);673 params.model = common_params_model{}; // make sure to clear model, so server starts in router mode674 });675 }676 677 // run all tasks in parallel678 if (!params.offline) {679 // if duplicated files are found, only download once (but still call on_done for each task)680 std::unordered_map<std::string, common_download_task *> unique_tasks;681 for (auto & task : tasks) {682 auto it = unique_tasks.find(task.local_path);683 if (it == unique_tasks.end()) {684 unique_tasks[task.local_path] = &task;685 }686 }687 std::vector<common_download_task> unique_tasks_vec;688 for (auto & pair : unique_tasks) {689 LOG_DBG("download task: %s -> %s\n", pair.second->url.c_str(), pair.second->local_path.c_str());690 unique_tasks_vec.push_back(*pair.second);691 }692 common_download_run_tasks(unique_tasks_vec);693 }694 695 // download successful, update params with the downloaded paths696 for (const auto & task : tasks) {697 if (task.on_done) {698 task.on_done();699 }700 }701}702 703//704// CLI argument parsing functions705//706 707static bool common_params_parse_ex(int argc, char ** argv, common_params_context & ctx_arg) {708 common_params & params = ctx_arg.params;709 710 // setup log directly from params.verbosity: see tools/cli/cli.cpp711 common_log_set_verbosity_thold(params.verbosity);712 713 std::unordered_map<std::string, std::pair<common_arg *, bool>> arg_to_options;714 for (auto & opt : ctx_arg.options) {715 for (const auto & arg : opt.args) {716 arg_to_options[arg] = {&opt, /* is_positive */ true};717 }718 for (const auto & arg : opt.args_neg) {719 arg_to_options[arg] = {&opt, /* is_positive */ false};720 }721 }722 723 // handle environment variables724 for (auto & opt : ctx_arg.options) {725 std::string value;726 if (opt.get_value_from_env(value)) {727 try {728 if (opt.handler_void && is_truthy(value)) {729 opt.handler_void(params);730 }731 if (opt.handler_int) {732 opt.handler_int(params, std::stoi(value));733 }734 if (opt.handler_bool) {735 opt.handler_bool(params, parse_bool_value(value));736 }737 if (opt.handler_string) {738 opt.handler_string(params, value);739 continue;740 }741 } catch (std::exception & e) {742 throw std::invalid_argument(string_format(743 "error while handling environment variable \"%s\": %s\n\n", opt.env, e.what()));744 }745 }746 }747 748 // handle command line arguments749 auto check_arg = [&](int i) {750 if (i+1 >= argc) {751 throw std::invalid_argument("expected value for argument");752 }753 };754 755 auto parse_cli_args = [&]() {756 std::set<std::string> seen_args;757 758 for (int i = 1; i < argc; i++) {759 const std::string arg_prefix = "--";760 761 std::string arg = argv[i];762 if (arg.compare(0, arg_prefix.size(), arg_prefix) == 0) {763 std::replace(arg.begin(), arg.end(), '_', '-');764 }765 if (arg_to_options.find(arg) == arg_to_options.end()) {766 throw std::invalid_argument(string_format("error: invalid argument: %s", arg.c_str()));767 }768 if (!seen_args.insert(arg).second) {769 const bool skip = (arg == "--spec-type");770 771 if (!skip) {772 LOG_WRN("DEPRECATED: argument '%s' specified multiple times, use comma-separated values instead (only last value will be used)\n", arg.c_str());773 }774 }775 auto & tmp = arg_to_options[arg];776 auto opt = *tmp.first;777 bool is_positive = tmp.second;778 if (opt.has_value_from_env()) {779 fprintf(stderr, "warn: %s environment variable is set, but will be overwritten by command line argument %s\n", opt.env, arg.c_str());780 }781 try {782 if (opt.handler_void) {783 opt.handler_void(params);784 continue;785 }786 if (opt.handler_bool) {787 opt.handler_bool(params, is_positive);788 continue;789 }790 791 // arg with single value792 check_arg(i);793 std::string val = argv[++i];794 if (opt.handler_int) {795 opt.handler_int(params, std::stoi(val));796 continue;797 }798 if (opt.handler_string) {799 opt.handler_string(params, val);800 continue;801 }802 803 // arg with 2 values804 check_arg(i);805 std::string val2 = argv[++i];806 if (opt.handler_str_str) {807 opt.handler_str_str(params, val, val2);808 continue;809 }810 } catch (std::exception & e) {811 throw std::invalid_argument(string_format(812 "error while handling argument \"%s\": %s\n\n"813 "usage:\n%s\n\nto show complete usage, run with -h",814 arg.c_str(), e.what(), opt.to_string().c_str()));815 }816 }817 818 // TODO: remove this check after deprecating --mmap|mlock|dio819 auto has_arg = [&](std::initializer_list<const char *> names) {820 return std::any_of(names.begin(), names.end(), [&](const char * name) {821 return seen_args.count(name);822 });823 };824 if (has_arg({"-lm", "--load-mode"}) &&825 has_arg({"--mlock", "--mmap", "--no-mmap", "-dio", "--direct-io", "-ndio", "--no-direct-io"})) {826 LOG_WRN("DEPRECATED: `--load-mode` and `--mlock`/`--mmap`/`--direct-io` should not be combined; only the last flag on the command line will take effect\n");827 }828 };829 830 // parse all CLI args now, so that -hf is available below for remote preset resolution831 parse_cli_args();832 833 postprocess_cpu_params(params.cpuparams, nullptr);834 postprocess_cpu_params(params.cpuparams_batch, ¶ms.cpuparams);835 836 postprocess_cpu_params(params.speculative.draft.cpuparams, ¶ms.cpuparams);837 postprocess_cpu_params(params.speculative.draft.cpuparams_batch, ¶ms.cpuparams_batch);838 839 if (params.prompt_cache_all && (params.interactive || params.interactive_first)) {840 throw std::invalid_argument("error: --prompt-cache-all not supported in interactive mode yet\n");841 }842 843 const bool skip_model_download =844 // server will call common_params_handle_models() later, so we skip it here845 ctx_arg.ex == LLAMA_EXAMPLE_SERVER ||846 // download calls common_params_handle_models() itself and prints the paths847 ctx_arg.ex == LLAMA_EXAMPLE_DOWNLOAD ||848 // export_graph_ops loads only metadata849 ctx_arg.ex == LLAMA_EXAMPLE_EXPORT_GRAPH_OPS;850 851 if (!skip_model_download) {852 // handle model and download853 common_models_handler handler = common_models_handler_init(params, ctx_arg.ex);854 common_models_handler_apply(handler, params);855 856 // model is required (except for server)857 // TODO @ngxson : maybe show a list of available models in CLI in this case858 bool can_skip_model = params.usage || params.completion || !params.server_base.empty();859 if (!can_skip_model && params.model.path.empty()) {860 throw std::invalid_argument("error: --model is required\n");861 }862 }863 864 if (params.escape) {865 string_process_escapes(params.prompt);866 string_process_escapes(params.input_prefix);867 string_process_escapes(params.input_suffix);868 for (auto & antiprompt : params.antiprompt) {869 string_process_escapes(antiprompt);870 }871 for (auto & seq_breaker : params.sampling.dry_sequence_breakers) {872 string_process_escapes(seq_breaker);873 }874 }875 876 if (!params.kv_overrides.empty()) {877 params.kv_overrides.emplace_back();878 params.kv_overrides.back().key[0] = 0;879 }880 881 const bool mcp_enabled = !params.mcp_servers_config.empty() || !params.mcp_servers_json.empty();882 if ((!params.server_tools.empty() || mcp_enabled) && !params.cors_origins_explicit) {883 LOG_WRN("server tools or MCP servers are enabled, using localhost as default CORS origin (change via --cors-origins)\n");884 params.cors_origins = "localhost";885 }886 887 // pad tensor_buft_overrides for llama_params_fit:888 const size_t ntbo = llama_max_tensor_buft_overrides();889 while (params.tensor_buft_overrides.size() < ntbo) {890 params.tensor_buft_overrides.push_back({nullptr, nullptr});891 }892 893 if (!params.speculative.draft.tensor_buft_overrides.empty()) {894 params.speculative.draft.tensor_buft_overrides.push_back({nullptr, nullptr});895 }896 897 if (!params.chat_template.empty() && !common_chat_verify_template(params.chat_template, params.use_jinja)) {898 throw std::runtime_error(string_format(899 "error: the supplied chat template is not supported: %s%s\n",900 params.chat_template.c_str(),901 params.use_jinja ? "" : "\nnote: llama.cpp was started without --jinja, we only support commonly used templates"902 ));903 }904 905 return true;906}907 908static void common_params_print_usage(common_params_context & ctx_arg) {909 auto print_options = [](std::vector<common_arg *> & options) {910 for (common_arg * opt : options) {911 printf("%s", opt->to_string().c_str());912 }913 };914 915 std::vector<common_arg *> common_options;916 std::vector<common_arg *> sampling_options;917 std::vector<common_arg *> spec_options;918 std::vector<common_arg *> specific_options;919 for (auto & opt : ctx_arg.options) {920 // in case multiple LLAMA_EXAMPLE_* are set, we prioritize the LLAMA_EXAMPLE_* matching current example921 if (opt.is_sampling) {922 sampling_options.push_back(&opt);923 } else if (opt.is_spec) {924 spec_options.push_back(&opt);925 } else if (opt.in_example(ctx_arg.ex)) {926 specific_options.push_back(&opt);927 } else {928 common_options.push_back(&opt);929 }930 }931 bool first = true;932 auto print_section = [&](const char * header, std::vector<common_arg *> & options) {933 if (options.empty()) {934 return;935 }936 printf("%s----- %s -----\n\n", first ? "" : "\n\n", header);937 first = false;938 print_options(options);939 };940 print_section("common params", common_options);941 print_section("sampling params", sampling_options);942 print_section("speculative params", spec_options);943 print_section("example-specific params", specific_options);944}945 946static void common_params_print_completion(common_params_context & ctx_arg) {947 std::vector<common_arg *> common_options;948 std::vector<common_arg *> sampling_options;949 std::vector<common_arg *> spec_options;950 std::vector<common_arg *> specific_options;951 952 for (auto & opt : ctx_arg.options) {953 if (opt.is_sampling) {954 sampling_options.push_back(&opt);955 } else if (opt.is_spec) {956 spec_options.push_back(&opt);957 } else if (opt.in_example(ctx_arg.ex)) {958 specific_options.push_back(&opt);959 } else {960 common_options.push_back(&opt);961 }962 }963 964 printf("_llama_completions() {\n");965 printf(" local cur prev opts\n");966 printf(" COMPREPLY=()\n");967 printf(" cur=\"${COMP_WORDS[COMP_CWORD]}\"\n");968 printf(" prev=\"${COMP_WORDS[COMP_CWORD-1]}\"\n\n");969 970 printf(" opts=\"");971 auto print_options = [](const std::vector<common_arg *> & options) {972 for (const common_arg * opt : options) {973 for (const char * arg : opt->args) {974 printf("%s ", arg);975 }976 }977 };978 979 print_options(common_options);980 print_options(sampling_options);981 print_options(spec_options);982 print_options(specific_options);983 printf("\"\n\n");984 985 printf(" case \"$prev\" in\n");986 printf(" --model|-m)\n");987 printf(" COMPREPLY=( $(compgen -f -X '!*.gguf' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");988 printf(" return 0\n");989 printf(" ;;\n");990 printf(" --grammar-file)\n");991 printf(" COMPREPLY=( $(compgen -f -X '!*.gbnf' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");992 printf(" return 0\n");993 printf(" ;;\n");994 printf(" --chat-template-file)\n");995 printf(" COMPREPLY=( $(compgen -f -X '!*.jinja' -- \"$cur\") $(compgen -d -- \"$cur\") )\n");996 printf(" return 0\n");997 printf(" ;;\n");998 printf(" *)\n");999 printf(" COMPREPLY=( $(compgen -W \"${opts}\" -- \"$cur\") )\n");1000 printf(" return 0\n");1001 printf(" ;;\n");1002 printf(" esac\n");1003 printf("}\n\n");1004 1005 std::set<std::string> executables = {1006 "llama-batched",1007 "llama-batched-bench",1008 "llama-bench",1009 "llama-cli",1010 "llama-completion",1011 "llama-convert-llama2c-to-ggml",1012 "llama-cvector-generator",1013 "llama-debug",1014 "llama-diffusion-cli",1015 "llama-embedding",1016 "llama-eval-callback",1017 "llama-export-lora",1018 "llama-finetune",1019 "llama-fit-params",1020 "llama-gemma3-cli",1021 "llama-gen-docs",1022 "llama-gguf",1023 "llama-gguf-hash",1024 "llama-gguf-split",1025 "llama-idle",1026 "llama-imatrix",1027 "llama-llava-cli",1028 "llama-lookahead",1029 "llama-lookup",1030 "llama-lookup-create",1031 "llama-lookup-merge",1032 "llama-lookup-stats",1033 "llama-minicpmv-cli",1034 "llama-mtmd-cli",1035 "llama-parallel",1036 "llama-passkey",1037 "llama-perplexity",1038 "llama-q8dot",1039 "llama-quantize",1040 "llama-qwen2vl-cli",1041 "llama-retrieval",1042 "llama-save-load-state",1043 "llama-server",1044 "llama-simple",1045 "llama-simple-chat",1046 "llama-speculative",1047 "llama-speculative-simple",1048 "llama-tokenize",1049 "llama-tts",1050 "llama-vdot"1051 };1052 1053 for (const auto& exe : executables) {1054 printf("complete -F _llama_completions %s\n", exe.c_str());1055 }1056}1057 1058static std::vector<ggml_backend_dev_t> parse_device_list(const std::string & value) {1059 std::vector<ggml_backend_dev_t> devices;1060 auto dev_names = string_split<std::string>(value, ',');1061 if (dev_names.empty()) {1062 throw std::invalid_argument("no devices specified");1063 }1064 if (dev_names.size() == 1 && dev_names[0] == "none") {1065 devices.push_back(nullptr);1066 } else {1067 ggml_backend_load_all();1068 for (const auto & device : dev_names) {1069 auto * dev = ggml_backend_dev_by_name(device.c_str());1070 if (!dev || ggml_backend_dev_type(dev) == GGML_BACKEND_DEVICE_TYPE_CPU) {1071 throw std::invalid_argument(string_format("invalid device: %s", device.c_str()));1072 }1073 devices.push_back(dev);1074 }1075 devices.push_back(nullptr);1076 }1077 return devices;1078}1079 1080void common_print_available_devices() {1081 constexpr size_t MiB = 1024 * 1024;1082 std::vector<ggml_backend_dev_t> devices;1083 1084 ggml_backend_load_all();1085 1086 for (size_t i = 0; i < ggml_backend_dev_count(); ++i) {1087 auto * dev = ggml_backend_dev_get(i);1088 if (ggml_backend_dev_type(dev) != GGML_BACKEND_DEVICE_TYPE_CPU) {1089 devices.push_back(dev);1090 }1091 }1092 printf("Available devices:\n");1093 1094 if (devices.empty()) {1095 printf(" (none)\n");1096 return;1097 }1098 for (auto * dev : devices) {1099 size_t free, total;1100 ggml_backend_dev_memory(dev, &free, &total);1101 printf(" %s: %s (%zu MiB, %zu MiB free)\n", ggml_backend_dev_name(dev), ggml_backend_dev_description(dev), total / MiB, free / MiB);1102 }1103}1104 1105static void add_rpc_devices(const std::string & servers) {1106 auto rpc_servers = string_split<std::string>(servers, ',');1107 if (rpc_servers.empty()) {1108 throw std::invalid_argument("no RPC servers specified");1109 }1110 ggml_backend_load_all();1111 ggml_backend_reg_t rpc_reg = ggml_backend_reg_by_name("RPC");1112 if (!rpc_reg) {1113 throw std::invalid_argument("failed to find RPC backend");1114 }1115 typedef ggml_backend_reg_t (*ggml_backend_rpc_add_server_t)(const char * endpoint);1116 ggml_backend_rpc_add_server_t ggml_backend_rpc_add_server_fn = (ggml_backend_rpc_add_server_t) ggml_backend_reg_get_proc_address(rpc_reg, "ggml_backend_rpc_add_server");1117 if (!ggml_backend_rpc_add_server_fn) {1118 throw std::invalid_argument("failed to find RPC add server function");1119 }1120 for (const auto & server : rpc_servers) {1121 auto reg = ggml_backend_rpc_add_server_fn(server.c_str());1122 ggml_backend_register(reg);1123 }1124}1125 1126bool common_params_to_map(int argc, char ** argv, llama_example ex, std::map<common_arg, std::string> & out_map) {1127 common_params dummy_params;1128 common_params_context ctx_arg = common_params_parser_init(dummy_params, ex, nullptr);1129 1130 std::unordered_map<std::string, common_arg *> arg_to_options;1131 for (auto & opt : ctx_arg.options) {1132 for (const auto & arg : opt.args) {1133 arg_to_options[arg] = &opt;1134 }1135 for (const auto & arg : opt.args_neg) {1136 arg_to_options[arg] = &opt;1137 }1138 }1139 1140 // TODO @ngxson : find a way to deduplicate this code1141 1142 // handle command line arguments1143 auto check_arg = [&](int i) {1144 if (i+1 >= argc) {1145 throw std::invalid_argument("expected value for argument");1146 }1147 };1148 1149 std::set<std::string> seen_args;1150 1151 for (int i = 1; i < argc; i++) {1152 const std::string arg_prefix = "--";1153 1154 std::string arg = argv[i];1155 if (arg.compare(0, arg_prefix.size(), arg_prefix) == 0) {1156 std::replace(arg.begin(), arg.end(), '_', '-');1157 }1158 if (arg_to_options.find(arg) == arg_to_options.end()) {1159 throw std::invalid_argument(string_format("error: invalid argument: %s", arg.c_str()));1160 }1161 if (!seen_args.insert(arg).second) {1162 const bool skip = (arg == "--spec-type");1163 1164 if (!skip) {1165 LOG_WRN("DEPRECATED: argument '%s' specified multiple times, use comma-separated values instead (only last value will be used)\n", arg.c_str());1166 }1167 }1168 auto opt = *arg_to_options[arg];1169 std::string val;1170 if (opt.value_hint == nullptr && opt.value_hint_2 == nullptr) {1171 // bool arg (need to reverse the meaning for negative args)1172 bool is_neg = std::find(opt.args_neg.begin(), opt.args_neg.end(), arg) != opt.args_neg.end();1173 val = is_neg ? "0" : "1";1174 }1175 if (opt.value_hint != nullptr) {1176 // arg with single value1177 check_arg(i);1178 val = argv[++i];1179 }1180 if (opt.value_hint_2 != nullptr) {1181 // TODO: support arg with 2 values1182 throw std::invalid_argument("error: argument with 2 values is not yet supported\n");1183 }1184 out_map[opt] = val;1185 }1186 1187 return true;1188}1189 1190#ifdef _WIN321191struct utf8_argv {1192 std::vector<std::string> buf;1193 std::vector<char*> ptrs;1194};1195 1196static utf8_argv make_utf8_argv() {1197 utf8_argv out;1198 int wargc = 0;1199 LPWSTR* wargv = CommandLineToArgvW(GetCommandLineW(), &wargc);1200 if (!wargv) return out;