Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#include "fit.h"2 3#include "log.h"4 5#include "../src/llama-ext.h"6 7#include <array>8#include <cassert>9#include <stdexcept>10#include <cinttypes>11#include <set>12#include <string>13#include <vector>14 15// this enum is only used in llama_params_fit_impl but needs to be defined outside of it to fix a Windows compilation issue16// enum to identify part of a layer for distributing its tensors:17enum common_layer_fraction_t {18 LAYER_FRACTION_NONE = 0, // nothing19 LAYER_FRACTION_ATTN = 1, // attention20 LAYER_FRACTION_UP = 2, // attention + up21 LAYER_FRACTION_GATE = 3, // attention + up + gate22 LAYER_FRACTION_MOE = 4, // everything but sparse MoE weights23};24 25class common_params_fit_exception : public std::runtime_error {26 using std::runtime_error::runtime_error;27};28 29static std::vector<llama_device_memory_data> common_get_device_memory_data_impl(30 const char * path_model,31 const llama_model_params * mparams,32 const llama_context_params * cparams,33 std::vector<ggml_backend_dev_t> & devs,34 uint32_t & hp_ngl,35 uint32_t & hp_n_ctx_train,36 uint32_t & hp_n_expert,37 ggml_log_level log_level) {38 struct user_data_t {39 struct {40 ggml_log_callback callback;41 void * user_data;42 } original_logger;43 ggml_log_level min_level; // prints below this log level go to debug log44 };45 user_data_t ud;46 llama_log_get(&ud.original_logger.callback, &ud.original_logger.user_data);47 ud.min_level = log_level;48 49 llama_log_set([](ggml_log_level level, const char * text, void * user_data) {50 const user_data_t * ud = (const user_data_t *) user_data;51 const ggml_log_level level_eff = level >= ud->min_level ? level : GGML_LOG_LEVEL_DEBUG;52 ud->original_logger.callback(level_eff, text, ud->original_logger.user_data);53 }, &ud);54 55 llama_model_params mparams_copy = *mparams;56 mparams_copy.no_alloc = true;57 mparams_copy.load_mode = LLAMA_LOAD_MODE_NONE;58 59 llama_model * model = llama_model_load_from_file(path_model, mparams_copy);60 if (model == nullptr) {61 llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);62 throw std::runtime_error("failed to load model");63 }64 65 llama_context * ctx = llama_init_from_model(model, *cparams);66 if (ctx == nullptr) {67 llama_model_free(model);68 llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);69 throw std::runtime_error("failed to create llama_context from model");70 }71 72 const size_t nd = llama_model_n_devices(model);73 std::vector<llama_device_memory_data> ret(nd + 1);74 75 llama_memory_breakdown memory_breakdown = llama_get_memory_breakdown(ctx);76 77 for (const auto & [buft, mb] : memory_breakdown) {78 if (ggml_backend_buft_is_host(buft)) {79 ret.back().mb.model += mb.model;80 ret.back().mb.context += mb.context;81 ret.back().mb.compute += mb.compute;82 continue;83 }84 85 ggml_backend_dev_t dev = ggml_backend_buft_get_device(buft);86 if (!dev) {87 continue;88 }89 for (size_t i = 0; i < nd; i++) {90 if (dev == llama_model_get_device(model, i)) {91 ret[i].mb.model += mb.model;92 ret[i].mb.context += mb.context;93 ret[i].mb.compute += mb.compute;94 break;95 }96 }97 }98 99 {100 ggml_backend_dev_t cpu_dev = ggml_backend_dev_by_type(GGML_BACKEND_DEVICE_TYPE_CPU);101 if (cpu_dev == nullptr) {102 throw std::runtime_error("no CPU backend found");103 }104 size_t free;105 size_t total;106 ggml_backend_dev_memory(cpu_dev, &free, &total);107 ret.back().free = free;108 ret.back().total = total;109 }110 for (size_t i = 0; i < nd; i++) {111 ggml_backend_dev_t dev = llama_model_get_device(model, i);112 113 size_t free;114 size_t total;115 ggml_backend_dev_memory(dev, &free, &total);116 117 // Some non-GPU accelerator backends, such as BLAS, report 0/0 and rely on118 // the host-memory fallback. For GPU-like backends, keep 0/0 so --fit does119 // not assign anything to a device with an unknown memory budget.120 if (free == 0 && total == 0) {121 const enum ggml_backend_dev_type type = ggml_backend_dev_type(dev);122 if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {123 LOG_WRN("%s: device %s did not report memory; --fit will not use it\n",124 __func__, ggml_backend_dev_name(dev));125 } else {126 free = ret.back().free;127 total = ret.back().total;128 }129 }130 ret[i].free = free;131 ret[i].total = total;132 }133 134 devs.clear();135 for (int i = 0; i < llama_model_n_devices(model); i++) {136 devs.push_back(llama_model_get_device(model, i));137 }138 139 hp_ngl = llama_model_n_layer(model);140 if (mparams->load_mtp) {141 hp_ngl += llama_model_n_layer_nextn(model);142 }143 hp_n_ctx_train = llama_model_n_ctx_train(model);144 hp_n_expert = llama_model_n_expert(model);145 146 common_memory_breakdown_print(ctx);147 148 llama_free(ctx);149 llama_model_free(model);150 llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);151 152 return ret;153}154 155common_device_memory_data_vec common_get_device_memory_data(156 const char * path_model,157 const llama_model_params * mparams,158 const llama_context_params * cparams,159 std::vector<ggml_backend_dev_t> & devs,160 uint32_t & hp_ngl,161 uint32_t & hp_n_ctx_train,162 uint32_t & hp_n_expert,163 ggml_log_level log_level) {164 std::vector<llama_device_memory_data> impl = common_get_device_memory_data_impl(165 path_model, mparams, cparams, devs, hp_ngl, hp_n_ctx_train, hp_n_expert, log_level);166 167 common_device_memory_data_vec ret(impl.size());168 for (size_t i = 0; i < impl.size(); i++) {169 ret[i].total = impl[i].total;170 ret[i].free = impl[i].free;171 ret[i].model = impl[i].mb.model;172 ret[i].context = impl[i].mb.context;173 ret[i].compute = impl[i].mb.compute;174 }175 return ret;176}177 178static void common_params_fit_impl(179 const char * path_model, struct llama_model_params * mparams, struct llama_context_params * cparams,180 float * tensor_split, struct llama_model_tensor_buft_override * tensor_buft_overrides,181 size_t * margins_s, uint32_t n_ctx_min, enum ggml_log_level log_level) {182 if (mparams->split_mode == LLAMA_SPLIT_MODE_TENSOR) {183 throw common_params_fit_exception("llama_params_fit is not implemented for SPLIT_MODE_TENSOR, abort");184 }185 constexpr int64_t MiB = 1024*1024;186 typedef std::vector<llama_device_memory_data> dmds_t;187 const llama_model_params default_mparams = llama_model_default_params();188 189 std::vector<ggml_backend_dev_t> devs;190 uint32_t hp_ngl = 0; // hparams.n_gpu_layers191 uint32_t hp_nct = 0; // hparams.n_ctx_train192 uint32_t hp_nex = 0; // hparams.n_expert193 194 // step 1: get data for default parameters and check whether any changes are necessary in the first place195 196 LOG_TRC("%s: getting device memory data for initial parameters:\n", __func__);197 const dmds_t dmds_full = common_get_device_memory_data_impl(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);198 const size_t nd = devs.size(); // number of devices199 200 std::vector<int64_t> margins; // this function uses int64_t rather than size_t for memory sizes to more conveniently handle deficits201 margins.reserve(nd);202 if (nd == 0) {203 margins.push_back(margins_s[0]);204 } else {205 for (size_t id = 0; id < nd; id++) {206 margins.push_back(margins_s[id]);207 }208 }209 210 std::vector<std::string> dev_names;211 {212 dev_names.reserve(nd);213 size_t max_length = 0;214 for (const auto & dev : devs) {215 std::string name = ggml_backend_dev_name(dev);216 name += " (";217 name += ggml_backend_dev_description(dev);218 name += ")";219 dev_names.push_back(name);220 max_length = std::max(max_length, name.length());221 }222 for (std::string & dn : dev_names) {223 dn.insert(dn.end(), max_length - dn.length(), ' ');224 }225 }226 227 int64_t sum_free = 0;228 int64_t sum_projected_free = 0;229 int64_t sum_projected_used = 0;230 int64_t sum_projected_model = 0;231 std::vector<int64_t> projected_free_per_device;232 projected_free_per_device.reserve(nd);233 234 if (nd == 0) {235 sum_projected_used = dmds_full.back().mb.total();236 sum_free = dmds_full.back().total;237 sum_projected_free = sum_free - sum_projected_used;238 LOG_TRC("%s: projected to use %" PRId64 " MiB of host memory vs. %" PRId64 " MiB of total host memory\n",239 __func__, sum_projected_used/MiB, sum_free/MiB);240 if (sum_projected_free >= margins[0]) {241 LOG_TRC("%s: will leave %" PRId64 " >= %" PRId64 " MiB of system memory, no changes needed\n",242 __func__, sum_projected_free/MiB, margins[0]/MiB);243 return;244 }245 } else {246 if (nd > 1) {247 LOG_TRC("%s: projected memory use with initial parameters [MiB]:\n", __func__);248 }249 for (size_t id = 0; id < nd; id++) {250 const llama_device_memory_data & dmd = dmds_full[id];251 252 const int64_t projected_used = dmd.mb.total();253 const int64_t projected_free = dmd.free - projected_used;254 projected_free_per_device.push_back(projected_free);255 256 sum_free += dmd.free;257 sum_projected_used += projected_used;258 sum_projected_free += projected_free;259 sum_projected_model += dmd.mb.model;260 261 if (nd > 1) {262 LOG_TRC("%s: - %s: %6" PRId64 " total, %6" PRId64 " used, %6" PRId64 " free vs. target of %6" PRId64 "\n",263 __func__, dev_names[id].c_str(), dmd.total/MiB, projected_used/MiB, projected_free/MiB, margins[id]/MiB);264 }265 }266 assert(sum_free >= 0 && sum_projected_used >= 0);267 LOG_TRC("%s: projected to use %" PRId64 " MiB of device memory vs. %" PRId64 " MiB of free device memory\n",268 __func__, sum_projected_used/MiB, sum_free/MiB);269 if (nd == 1) {270 if (projected_free_per_device[0] >= margins[0]) {271 LOG_TRC("%s: will leave %" PRId64 " >= %" PRId64 " MiB of free device memory, no changes needed\n",272 __func__, projected_free_per_device[0]/MiB, margins[0]/MiB);273 return;274 }275 } else {276 bool changes_needed = false;277 for (size_t id = 0; id < nd; id++) {278 if (projected_free_per_device[id] < margins[id]) {279 changes_needed = true;280 break;281 }282 }283 if (!changes_needed) {284 LOG_TRC("%s: targets for free memory can be met on all devices, no changes needed\n", __func__);285 return;286 }287 }288 }289 290 // step 2: try reducing memory use by reducing the context size291 292 {293 int64_t global_surplus = sum_projected_free;294 if (nd == 0) {295 global_surplus -= margins[0];296 } else {297 for (size_t id = 0; id < nd; id++) {298 global_surplus -= margins[id];299 }300 }301 if (global_surplus < 0) {302 if (nd <= 1) {303 LOG_TRC("%s: cannot meet free memory target of %" PRId64 " MiB, need to reduce device memory by %" PRId64 " MiB\n",304 __func__, margins[0]/MiB, -global_surplus/MiB);305 } else {306 LOG_TRC(307 "%s: cannot meet free memory targets on all devices, need to use %" PRId64 " MiB less in total\n",308 __func__, -global_surplus/MiB);309 }310 if (cparams->n_ctx == 0) {311 if (hp_nct > n_ctx_min) {312 int64_t sum_used_target = sum_free;313 if (nd == 0) {314 sum_used_target -= margins[0];315 } else {316 for (size_t id = 0; id < nd; id++) {317 sum_used_target -= margins[id];318 }319 }320 if (nd > 1) {321 // for multiple devices we need to be more conservative in terms of how much context we think can fit:322 // - for dense models only whole layers can be assigned to devices323 // - for MoE models only whole tensors can be assigned to devices, which we estimate to be <= 1/3 of a layer324 // - on average we expect a waste of 0.5 layers/tensors per device325 // - use slightly more than the expected average for nd devices to be safe326 const int64_t model_per_layer = sum_projected_model / std::min(uint32_t(mparams->n_gpu_layers), hp_ngl);327 sum_used_target -= (nd + 1) * model_per_layer / (hp_nex == 0 ? 2 : 6);328 }329 330 int64_t sum_projected_used_min_ctx = 0;331 cparams->n_ctx = n_ctx_min;332 const dmds_t dmds_min_ctx = common_get_device_memory_data_impl(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);333 if (nd == 0) {334 sum_projected_used_min_ctx = dmds_min_ctx.back().mb.total();335 } else {336 for (size_t id = 0; id < nd; id++) {337 sum_projected_used_min_ctx += dmds_min_ctx[id].mb.total();338 }339 }340 if (sum_used_target > sum_projected_used_min_ctx) {341 // linear interpolation between minimum and maximum context size:342 cparams->n_ctx += (hp_nct - n_ctx_min) * (sum_used_target - sum_projected_used_min_ctx)343 / (sum_projected_used - sum_projected_used_min_ctx);344 cparams->n_ctx = std::max(cparams->n_ctx - cparams->n_ctx % 256, n_ctx_min); // round down context for CUDA backend345 346 const int64_t bytes_per_ctx = (sum_projected_used - sum_projected_used_min_ctx) / (hp_nct - n_ctx_min);347 const int64_t memory_reduction = (hp_nct - cparams->n_ctx) * bytes_per_ctx;348 LOG_TRC("%s: context size reduced from %" PRIu32 " to %" PRIu32 " -> need %" PRId64 " MiB less memory in total\n",349 __func__, hp_nct, cparams->n_ctx, memory_reduction/MiB);350 if (nd <= 1) {351 LOG_TRC("%s: entire model can be fit by reducing context\n", __func__);352 return;353 }354 LOG_TRC("%s: entire model should be fit across devices by reducing context\n", __func__);355 } else {356 const int64_t memory_reduction = sum_projected_used - sum_projected_used_min_ctx;357 LOG_TRC("%s: context size reduced from %" PRIu32 " to %" PRIu32 " -> need %" PRId64 " MiB less memory in total\n",358 __func__, hp_nct, cparams->n_ctx, memory_reduction/MiB);359 }360 } else {361 if (n_ctx_min == UINT32_MAX) {362 LOG_TRC("%s: user has requested full context size of %" PRIu32 " -> no change\n", __func__, hp_nct);363 } else {364 LOG_TRC("%s: default model context size is %" PRIu32 " which is <= the min. context size of %" PRIu32 " -> no change\n",365 __func__, hp_nct, n_ctx_min);366 }367 }368 } else {369 LOG_TRC("%s: context size set by user to %" PRIu32 " -> no change\n", __func__, cparams->n_ctx);370 }371 }372 }373 if (nd == 0) {374 throw common_params_fit_exception("was unable to fit model into system memory by reducing context, abort");375 }376 377 if (mparams->n_gpu_layers != default_mparams.n_gpu_layers) {378 throw common_params_fit_exception("n_gpu_layers already set by user to " + std::to_string(mparams->n_gpu_layers) + ", abort");379 }380 if (nd > 1) {381 if (!tensor_split) {382 throw common_params_fit_exception("did not provide a buffer to write the tensor_split to, abort");383 }384 if (mparams->tensor_split) {385 for (size_t id = 0; id < nd; id++) {386 if (mparams->tensor_split[id] != 0.0f) {387 throw common_params_fit_exception("model_params::tensor_split already set by user, abort");388 }389 }390 }391 if (mparams->split_mode == LLAMA_SPLIT_MODE_ROW) {392 throw common_params_fit_exception("changing weight allocation for LLAMA_SPLIT_MODE_ROW not implemented, abort");393 }394 }395 if (!tensor_buft_overrides) {396 throw common_params_fit_exception("did not provide buffer to set tensor_buft_overrides, abort");397 }398 if (mparams->tensor_buft_overrides && (mparams->tensor_buft_overrides->pattern || mparams->tensor_buft_overrides->buft)) {399 throw common_params_fit_exception("model_params::tensor_buft_overrides already set by user, abort");400 }401 402 // step 3: iteratively fill the back to front with "dense" layers403 // - for a dense model simply fill full layers, giving each device a contiguous slice of the model404 // - for a MoE model, same as dense model but with all MoE tensors in system memory405 406 // utility function that returns a static C string matching the tensors for a specific layer index and layer fraction:407 auto get_overflow_pattern = [&](const size_t il, const common_layer_fraction_t lf) -> const char * {408 constexpr size_t n_strings = 1000;409 if (il >= n_strings) {410 throw std::runtime_error("at most " + std::to_string(n_strings) + " model layers are supported");411 }412 switch (lf) {413 case LAYER_FRACTION_ATTN: {414 static std::array<std::string, n_strings> patterns;415 if (patterns[il].empty()) {416 patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(gate|up|gate_up|down).*";417 }418 return patterns[il].c_str();419 }420 case LAYER_FRACTION_UP: {421 static std::array<std::string, n_strings> patterns;422 if (patterns[il].empty()) {423 patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(gate|gate_up|down).*";424 }425 return patterns[il].c_str();426 }427 case LAYER_FRACTION_GATE: {428 static std::array<std::string, n_strings> patterns;429 if (patterns[il].empty()) {430 patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_down.*";431 }432 return patterns[il].c_str();433 }434 case LAYER_FRACTION_MOE: {435 static std::array<std::string, n_strings> patterns;436 if (patterns[il].empty()) {437 patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(up|down|gate_up|gate)_(ch|)exps";438 }439 return patterns[il].c_str();440 }441 default:442 GGML_ABORT("fatal error");443 }444 };445 446 struct ngl_t {447 uint32_t n_layer = 0; // number of total layers448 uint32_t n_part = 0; // number of partial layers, <= n_layer449 450 // for the first partial layer varying parts can overflow, all further layers use LAYER_FRACTION_MOE:451 common_layer_fraction_t overflow_type = LAYER_FRACTION_MOE;452 453 uint32_t n_full() const {454 assert(n_layer >= n_part);455 return n_layer - n_part;456 }457 };458 459 const size_t ntbo = llama_max_tensor_buft_overrides();460 461 // utility function to set n_gpu_layers and tensor_split462 auto set_ngl_tensor_split_tbo = [&](463 const std::vector<ngl_t> & ngl_per_device,464 const std::vector<ggml_backend_buffer_type_t> & overflow_bufts,465 llama_model_params & mparams) {466 mparams.n_gpu_layers = 0;467 for (size_t id = 0; id < nd; id++) {468 mparams.n_gpu_layers += ngl_per_device[id].n_layer;469 if (nd > 1) {470 tensor_split[id] = ngl_per_device[id].n_layer;471 }472 }473 assert(uint32_t(mparams.n_gpu_layers) <= hp_ngl + 1);474 uint32_t il0 = hp_ngl + 1 - mparams.n_gpu_layers; // start index for tensor buft overrides475 476 mparams.tensor_split = tensor_split;477 478 size_t itbo = 0;479 for (size_t id = 0; id < nd; id++) {480 il0 += ngl_per_device[id].n_full();481 for (uint32_t il = il0; il < il0 + ngl_per_device[id].n_part; il++) {482 if (itbo + 1 >= ntbo) {483 tensor_buft_overrides[itbo].pattern = nullptr;484 tensor_buft_overrides[itbo].buft = nullptr;485 itbo++;486 mparams.tensor_buft_overrides = tensor_buft_overrides;487 throw common_params_fit_exception("llama_max_tensor_buft_overrides() == "488 + std::to_string(ntbo) + " is insufficient for model");489 }490 tensor_buft_overrides[itbo].pattern = get_overflow_pattern(il, il == il0 ? ngl_per_device[id].overflow_type : LAYER_FRACTION_MOE);491 tensor_buft_overrides[itbo].buft = il == il0 ? overflow_bufts[id] : ggml_backend_cpu_buffer_type();492 itbo++;493 }494 il0 += ngl_per_device[id].n_part;495 }496 tensor_buft_overrides[itbo].pattern = nullptr;497 tensor_buft_overrides[itbo].buft = nullptr;498 itbo++;499 mparams.tensor_buft_overrides = tensor_buft_overrides;500 };501 502 // utility function that returns the memory use per device for given numbers of layers per device503 auto get_memory_for_layers = [&](504 const char * func_name,505 const std::vector<ngl_t> & ngl_per_device,506 const std::vector<ggml_backend_buffer_type_t> & overflow_bufts) -> std::vector<int64_t> {507 llama_model_params mparams_copy = *mparams;508 set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, mparams_copy);509 510 const dmds_t dmd_nl = common_get_device_memory_data_impl(511 path_model, &mparams_copy, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);512 513 LOG_TRC("%s: memory for test allocation by device:\n", func_name);514 for (size_t id = 0; id < nd; id++) {515 const ngl_t & n = ngl_per_device[id];516 LOG_TRC(517 "%s: id=%zu, n_layer=%2" PRIu32 ", n_part=%2" PRIu32 ", overflow_type=%d, mem=%6" PRId64 " MiB\n",518 func_name, id, n.n_layer, n.n_part, int(n.overflow_type), dmd_nl[id].mb.total()/MiB);519 }520 521 std::vector<int64_t> ret;522 ret.reserve(nd);523 for (size_t id = 0; id < nd; id++) {524 ret.push_back(dmd_nl[id].mb.total());525 }526 return ret;527 };528 529 int64_t global_surplus_cpu_moe = 0;530 if (hp_nex > 0) {531 const static std::string pattern_moe_all = "blk\\.\\d+\\.ffn_(up|down|gate_up|gate)_(ch|)exps"; // matches all MoE tensors532 ggml_backend_buffer_type_t cpu_buft = ggml_backend_cpu_buffer_type();533 tensor_buft_overrides[0] = {pattern_moe_all.c_str(), cpu_buft};534 tensor_buft_overrides[1] = {nullptr, nullptr};535 mparams->tensor_buft_overrides = tensor_buft_overrides;536 537 LOG_TRC("%s: getting device memory data with all MoE tensors moved to system memory:\n", __func__);538 const dmds_t dmds_cpu_moe = common_get_device_memory_data_impl(539 path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);540 541 for (size_t id = 0; id < nd; id++) {542 global_surplus_cpu_moe += dmds_cpu_moe[id].free;543 global_surplus_cpu_moe -= int64_t(dmds_cpu_moe[id].mb.total()) + margins[id];544 }545 546 if (global_surplus_cpu_moe > 0) {547 LOG_TRC("%s: with only dense weights in device memory there is a total surplus of %" PRId64 " MiB\n",548 __func__, global_surplus_cpu_moe/MiB);549 } else {550 LOG_TRC("%s: with only dense weights in device memory there is still a total deficit of %" PRId64 " MiB\n",551 __func__, -global_surplus_cpu_moe/MiB);552 }553 554 // reset555 tensor_buft_overrides[0] = {nullptr, nullptr};556 mparams->tensor_buft_overrides = tensor_buft_overrides;557 }558 559 std::vector<int64_t> targets; // maximum acceptable memory use per device560 targets.reserve(nd);561 for (size_t id = 0; id < nd; id++) {562 targets.push_back(dmds_full[id].free - margins[id]);563 LOG_TRC("%s: id=%zu, target=%" PRId64 " MiB\n", __func__, id, targets[id]/MiB);564 }565 566 std::vector<ggml_backend_buffer_type_t> overflow_bufts; // which bufts the first partial layer of a device overflows to:567 overflow_bufts.reserve(nd);568 for (size_t id = 0; id < nd; id++) {569 overflow_bufts.push_back(ggml_backend_cpu_buffer_type());570 }571 572 std::vector<ngl_t> ngl_per_device(nd);573 std::vector<int64_t> mem = get_memory_for_layers(__func__, ngl_per_device, overflow_bufts);574 575 // optimize the number of layers per device using the method of false position:576 // - ngl_per_device has 0 layers for each device, lower bound577 // - try a "high" configuration where a device is given all unassigned layers578 // - interpolate the memory use / layer between low and high linearly to get a guess where it meets our target579 // - check memory use of our guess, replace either the low or high bound580 // - once we only have a difference of a single layer, stop and return the lower bound that just barely still fits581 // - the last device has the output layer, which cannot be a partial layer582 if (hp_nex == 0) {583 LOG_TRC("%s: filling dense layers back-to-front:\n", __func__);584 } else {585 LOG_TRC("%s: filling dense-only layers back-to-front:\n", __func__);586 }587 for (int id = nd - 1; id >= 0; id--) {588 uint32_t n_unassigned = hp_ngl + 1;589 for (size_t jd = id + 1; jd < nd; ++jd) {590 assert(n_unassigned >= ngl_per_device[jd].n_layer);591 n_unassigned -= ngl_per_device[jd].n_layer;592 }593 594 std::vector<ngl_t> ngl_per_device_high = ngl_per_device;595 ngl_per_device_high[id].n_layer = n_unassigned;596 if (hp_nex > 0) {597 ngl_per_device_high[id].n_part = size_t(id) < nd - 1 ? ngl_per_device_high[id].n_layer : ngl_per_device_high[id].n_layer - 1;598 }599 if (ngl_per_device_high[id].n_layer > 0) {600 std::vector<int64_t> mem_high = get_memory_for_layers(__func__, ngl_per_device_high, overflow_bufts);601 if (mem_high[id] > targets[id]) {602 assert(ngl_per_device_high[id].n_layer > ngl_per_device[id].n_layer);603 uint32_t delta = ngl_per_device_high[id].n_layer - ngl_per_device[id].n_layer;604 LOG_TRC("%s: start filling device %" PRIu32 ", delta=%" PRIu32 "\n", __func__, id, delta);605 while (delta > 1) {606 uint32_t step_size = int64_t(delta) * (targets[id] - mem[id]) / (mem_high[id] - mem[id]);607 step_size = std::max(step_size, uint32_t(1));608 step_size = std::min(step_size, delta - 1);609 610 std::vector<ngl_t> ngl_per_device_test = ngl_per_device;611 ngl_per_device_test[id].n_layer += step_size;612 if (hp_nex) {613 ngl_per_device_test[id].n_part += size_t(id) == nd - 1 && ngl_per_device_test[id].n_part == 0 ?614 step_size - 1 : step_size; // the first layer is the output layer which must always be full615 }616 const std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts);617 618 if (mem_test[id] <= targets[id]) {619 ngl_per_device = ngl_per_device_test;620 mem = mem_test;621 LOG_TRC("%s: set ngl_per_device[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device[id].n_layer);622 } else {623 ngl_per_device_high = ngl_per_device_test;624 mem_high = mem_test;625 LOG_TRC("%s: set ngl_per_device_high[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device_high[id].n_layer);626 }627 delta = ngl_per_device_high[id].n_layer - ngl_per_device[id].n_layer;628 }629 } else {630 assert(ngl_per_device_high[id].n_layer == n_unassigned);631 ngl_per_device = ngl_per_device_high;632 mem = mem_high;633 LOG_TRC("%s: set ngl_per_device[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device[id].n_layer);634 }635 }636 637 const int64_t projected_margin = dmds_full[id].free - mem[id];638 LOG_TRC(639 "%s: - %s: %2" PRIu32 " layers, %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",640 __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, mem[id]/MiB, projected_margin/MiB);641 }642 if (hp_nex == 0 || global_surplus_cpu_moe <= 0) {643 set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, *mparams);644 return;645 }646 647 // step 4: for a MoE model where all dense tensors fit,648 // convert the dense-only layers in the back to full layers in the front until all devices are full649 // essentially the same procedure as for the dense-only layers except front-to-back650 // also, try fitting at least part of one more layer to reduce waste for "small" GPUs with e.g. 24 GiB VRAM651 652 size_t id_dense_start = nd;653 for (int id = nd - 1; id >= 0; id--) {654 if (ngl_per_device[id].n_layer > 0) {655 id_dense_start = id;656 continue;657 }658 break;659 }660 assert(id_dense_start < nd);661 662 LOG_TRC("%s: converting dense-only layers to full layers and filling them front-to-back with overflow to next device/system memory:\n", __func__);663 for (size_t id = 0; id <= id_dense_start && id_dense_start < nd; id++) {664 std::vector<ngl_t> ngl_per_device_high = ngl_per_device;665 for (size_t jd = id_dense_start; jd < nd; jd++) {666 const uint32_t n_layer_move = jd < nd - 1 ? ngl_per_device_high[jd].n_layer : ngl_per_device_high[jd].n_layer - 1;667 ngl_per_device_high[id].n_layer += n_layer_move;668 ngl_per_device_high[jd].n_layer -= n_layer_move;669 ngl_per_device_high[jd].n_part = 0;670 }671 size_t id_dense_start_high = nd - 1;672 std::vector<int64_t> mem_high = get_memory_for_layers(__func__, ngl_per_device_high, overflow_bufts);673 674 if (mem_high[id] > targets[id]) {675 assert(ngl_per_device_high[id].n_full() >= ngl_per_device[id].n_full());676 uint32_t delta = ngl_per_device_high[id].n_full() - ngl_per_device[id].n_full();677 while (delta > 1) {678 uint32_t step_size = int64_t(delta) * (targets[id] - mem[id]) / (mem_high[id] - mem[id]);679 step_size = std::max(step_size, uint32_t(1));680 step_size = std::min(step_size, delta - 1);681 682 std::vector<ngl_t> ngl_per_device_test = ngl_per_device;683 size_t id_dense_start_test = id_dense_start;684 uint32_t n_converted_test = 0;685 for (;id_dense_start_test < nd; id_dense_start_test++) {686 const uint32_t n_convert_jd = std::min(step_size - n_converted_test, ngl_per_device_test[id_dense_start_test].n_part);687 ngl_per_device_test[id_dense_start_test].n_layer -= n_convert_jd;688 ngl_per_device_test[id_dense_start_test].n_part -= n_convert_jd;689 ngl_per_device_test[id].n_layer += n_convert_jd;690 n_converted_test += n_convert_jd;691 692 if (ngl_per_device_test[id_dense_start_test].n_part > 0) {693 break;694 }695 }696 const std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts);697 698 if (mem_test[id] <= targets[id]) {699 ngl_per_device = ngl_per_device_test;700 mem = mem_test;701 id_dense_start = id_dense_start_test;702 LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start=%zu\n",703 __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);704 } else {705 ngl_per_device_high = ngl_per_device_test;706 mem_high = mem_test;707 id_dense_start_high = id_dense_start_test;708 LOG_TRC("%s: set ngl_per_device_high[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start_high=%zu\n",709 __func__, id, ngl_per_device_high[id].n_layer, ngl_per_device_high[id].n_part, id_dense_start_high);710 }711 assert(ngl_per_device_high[id].n_full() >= ngl_per_device[id].n_full());712 delta = ngl_per_device_high[id].n_full() - ngl_per_device[id].n_full();713 }714 } else {715 ngl_per_device = ngl_per_device_high;716 mem = mem_high;717 id_dense_start = id_dense_start_high;718 LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start=%zu\n",719 __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);720 }721 722 // try to fit at least part of one more layer723 if (ngl_per_device[id_dense_start].n_layer > (id < nd - 1 ? 0 : 1)) {724 std::vector<ngl_t> ngl_per_device_test = ngl_per_device;725 size_t id_dense_start_test = id_dense_start;726 ngl_per_device_test[id_dense_start_test].n_layer--;727 ngl_per_device_test[id_dense_start_test].n_part--;728 ngl_per_device_test[id].n_layer++;729 ngl_per_device_test[id].n_part++;730 if (ngl_per_device_test[id_dense_start_test].n_part == 0) {731 id_dense_start_test++;732 }733 ngl_per_device_test[id].overflow_type = LAYER_FRACTION_UP;734 std::vector<ggml_backend_buffer_type_t> overflow_bufts_test = overflow_bufts;735 if (id < nd - 1) {736 overflow_bufts_test[id] = ggml_backend_dev_buffer_type(devs[id + 1]);737 }738 LOG_TRC("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_UP\n", __func__);739 std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);740 if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {741 ngl_per_device = ngl_per_device_test;742 overflow_bufts = overflow_bufts_test;743 mem = mem_test;744 id_dense_start = id_dense_start_test;745 LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", UP), id_dense_start=%zu\n",746 __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);747 748 ngl_per_device_test[id].overflow_type = LAYER_FRACTION_GATE;749 LOG_TRC("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_GATE\n", __func__);750 mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);751 if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {752 ngl_per_device = ngl_per_device_test;753 overflow_bufts = overflow_bufts_test;754 mem = mem_test;755 id_dense_start = id_dense_start_test;756 LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", GATE), id_dense_start=%zu\n",757 __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);758 }759 } else {760 ngl_per_device_test[id].overflow_type = LAYER_FRACTION_ATTN;761 LOG_TRC("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_ATTN\n", __func__);762 mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);763 if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {764 ngl_per_device = ngl_per_device_test;765 overflow_bufts = overflow_bufts_test;766 mem = mem_test;767 id_dense_start = id_dense_start_test;768 LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", ATTN), id_dense_start=%zu\n",769 __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);770 }771 }772 }773 774 const int64_t projected_margin = dmds_full[id].free - mem[id];775 LOG_TRC(776 "%s: - %s: %2" PRIu32 " layers (%2" PRIu32 " overflowing), %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",777 __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, ngl_per_device[id].n_part, mem[id]/MiB, projected_margin/MiB);778 }779 780 // print info for devices that were not changed during the conversion from dense only to full layers:781 for (size_t id = id_dense_start + 1; id < nd; id++) {782 const int64_t projected_margin = dmds_full[id].free - mem[id];783 LOG_TRC(784 "%s: - %s: %2" PRIu32 " layers (%2" PRIu32 " overflowing), %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",785 __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, ngl_per_device[id].n_part, mem[id]/MiB, projected_margin/MiB);786 }787 788 set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, *mparams);789}790 791enum common_params_fit_status common_fit_params(792 const char * path_model,793 llama_model_params * mparams,794 llama_context_params * cparams,795 float * tensor_split,796 llama_model_tensor_buft_override * tensor_buft_overrides,797 size_t * margins,798 uint32_t n_ctx_min,799 ggml_log_level log_level) {800 const int64_t t0_us = llama_time_us();801 common_params_fit_status status = COMMON_PARAMS_FIT_STATUS_SUCCESS;802 try {803 common_params_fit_impl(path_model, mparams, cparams, tensor_split, tensor_buft_overrides, margins, n_ctx_min, log_level);804 LOG_TRC("%s: successfully fit params to free device memory\n", __func__);805 } catch (const common_params_fit_exception & e) {806 LOG_WRN("%s: failed to fit params to free device memory: %s\n", __func__, e.what());807 status = COMMON_PARAMS_FIT_STATUS_FAILURE;808 } catch (const std::runtime_error & e) {809 LOG_ERR("%s: encountered an error while trying to fit params to free device memory: %s\n", __func__, e.what());810 status = COMMON_PARAMS_FIT_STATUS_ERROR;811 }812 const int64_t t1_us = llama_time_us();813 LOG_TRC("%s: fitting params to free memory took %.2f seconds\n", __func__, (t1_us - t0_us) * 1e-6);814 return status;815}816 817void common_memory_breakdown_print(const struct llama_context * ctx) {818 //const auto & devices = ctx->get_model().devices;819 const auto * model = llama_get_model(ctx);820 821 std::vector<ggml_backend_dev_t> devices;822 for (int i = 0; i < llama_model_n_devices(model); i++) {823 devices.push_back(llama_model_get_device(model, i));824 }825 826 llama_memory_breakdown memory_breakdown = llama_get_memory_breakdown(ctx);827 828 std::vector<std::array<std::string, 9>> table_data;829 table_data.reserve(devices.size());830 const std::string template_header = "%s: | %s | %s %s %s %s %s %s %s |\n";831 const std::string template_gpu = "%s: | %s | %s = %s + (%s = %s + %s + %s) + %s |\n";832 const std::string template_other = "%s: | %s | %s %s %s = %s + %s + %s %s |\n";833 834 table_data.push_back({template_header, "memory breakdown [MiB]", "total", "free", "self", "model", "context", "compute", "unaccounted"});835 836 constexpr size_t MiB = 1024 * 1024;837 const std::vector<std::string> desc_prefixes_strip = {"NVIDIA ", "GeForce ", "Tesla ", "AMD ", "Radeon ", "Instinct "};838 839 // track seen buffer types to avoid double counting:840 std::set<ggml_backend_buffer_type_t> seen_buffer_types;841 842 // accumulative memory breakdown for each device and for host:843 std::vector<llama_memory_breakdown_data> mb_dev(devices.size());844 llama_memory_breakdown_data mb_host;845 846 for (const auto & buft_mb : memory_breakdown) {847 ggml_backend_buffer_type_t buft = buft_mb.first;848 const llama_memory_breakdown_data & mb = buft_mb.second;849 if (ggml_backend_buft_is_host(buft)) {850 mb_host.model += mb.model;851 mb_host.context += mb.context;852 mb_host.compute += mb.compute;853 seen_buffer_types.insert(buft);854 continue;855 }856 ggml_backend_dev_t dev = ggml_backend_buft_get_device(buft);857 if (dev) {858 int i_dev = -1;859 for (size_t i = 0; i < devices.size(); i++) {860 if (devices[i] == dev) {861 i_dev = i;862 break;863 }864 }865 if (i_dev != -1) {866 mb_dev[i_dev].model += mb.model;867 mb_dev[i_dev].context += mb.context;868 mb_dev[i_dev].compute += mb.compute;869 seen_buffer_types.insert(buft);870 continue;871 }872 }873 }874 875 // print memory breakdown for each device:876 for (size_t i = 0; i < devices.size(); i++) {877 ggml_backend_dev_t dev = devices[i];878 llama_memory_breakdown_data mb = mb_dev[i];879 880 const std::string name = ggml_backend_dev_name(dev);881 std::string desc = ggml_backend_dev_description(dev);882 for (const std::string & prefix : desc_prefixes_strip) {883 if (desc.length() >= prefix.length() && desc.substr(0, prefix.length()) == prefix) {884 desc = desc.substr(prefix.length());885 }886 }887 888 size_t free, total;889 ggml_backend_dev_memory(dev, &free, &total);890 891 const size_t self = mb.model + mb.context + mb.compute;892 const int64_t unaccounted = static_cast<int64_t>(total) - static_cast<int64_t>(free) - static_cast<int64_t>(self);893 894 table_data.push_back({895 template_gpu,896 " - " + name + " (" + desc + ")",897 std::to_string(total / MiB),898 std::to_string(free / MiB),899 std::to_string(self / MiB),900 std::to_string(mb.model / MiB),901 std::to_string(mb.context / MiB),902 std::to_string(mb.compute / MiB),903 std::to_string(unaccounted / static_cast<int64_t>(MiB))});904 }905 906 // print memory breakdown for host:907 {908 const size_t self = mb_host.model + mb_host.context + mb_host.compute;909 table_data.push_back({910 template_other,911 " - Host",912 "", // total913 "", // free914 std::to_string(self / MiB),915 std::to_string(mb_host.model / MiB),916 std::to_string(mb_host.context / MiB),917 std::to_string(mb_host.compute / MiB),918 ""}); // unaccounted919 }920 921 // print memory breakdown for all remaining buffer types:922 for (const auto & buft_mb : memory_breakdown) {923 ggml_backend_buffer_type_t buft = buft_mb.first;924 const llama_memory_breakdown_data & mb = buft_mb.second;925 if (seen_buffer_types.count(buft) == 1) {926 continue;927 }928 const std::string name = ggml_backend_buft_name(buft);929 const size_t self = mb.model + mb.context + mb.compute;930 table_data.push_back({931 template_other,932 " - " + name,933 "", // total934 "", // free935 std::to_string(self / MiB),936 std::to_string(mb.model / MiB),937 std::to_string(mb.context / MiB),938 std::to_string(mb.compute / MiB),939 ""}); // unaccounted940 seen_buffer_types.insert(buft);941 }942 943 for (size_t j = 1; j < table_data[0].size(); j++) {944 size_t max_len = 0;945 for (const auto & td : table_data) {946 max_len = std::max(max_len, td[j].length());947 }948 for (auto & td : table_data) {949 td[j].insert(j == 1 ? td[j].length() : 0, max_len - td[j].length(), ' ');950 }951 }952 for (const auto & td : table_data) {953 LOG_TRC(td[0].c_str(),954 __func__, td[1].c_str(), td[2].c_str(), td[3].c_str(), td[4].c_str(), td[5].c_str(),955 td[6].c_str(), td[7].c_str(), td[8].c_str());956 }957}958 959void common_fit_print(960 const char * path_model,961 llama_model_params * mparams,962 llama_context_params * cparams) {963 std::vector<ggml_backend_dev_t> devs;964 uint32_t hp_ngl = 0; // hparams.n_gpu_layers965 uint32_t hp_nct = 0; // hparams.n_ctx_train966 uint32_t hp_nex = 0; // hparams.n_expert967 968 auto dmd = common_get_device_memory_data_impl(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, GGML_LOG_LEVEL_ERROR);969 GGML_ASSERT(dmd.size() == devs.size() + 1);970 971 for (size_t id = 0; id < devs.size(); id++) {972 printf("%s ", ggml_backend_dev_name(devs[id]));973 printf("%zu ", dmd[id].mb.model/1024/1024);974 printf("%zu ", dmd[id].mb.context/1024/1024);975 printf("%zu ", dmd[id].mb.compute/1024/1024);976 printf("\n");977 }978 979 printf("Host ");980 printf("%zu ", dmd.back().mb.model/1024/1024);981 printf("%zu ", dmd.back().mb.context/1024/1024);982 printf("%zu ", dmd.back().mb.compute/1024/1024);983 printf("\n");984}985 