Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
fit.cpp985 linesDownload Raw Back to common
1#include "fit.h"2 3#include "log.h"4 5#include "../src/llama-ext.h"6 7#include <array>8#include <cassert>9#include <stdexcept>10#include <cinttypes>11#include <set>12#include <string>13#include <vector>14 15// this enum is only used in llama_params_fit_impl but needs to be defined outside of it to fix a Windows compilation issue16// enum to identify part of a layer for distributing its tensors:17enum common_layer_fraction_t {18    LAYER_FRACTION_NONE = 0, // nothing19    LAYER_FRACTION_ATTN = 1, // attention20    LAYER_FRACTION_UP   = 2, // attention + up21    LAYER_FRACTION_GATE = 3, // attention + up + gate22    LAYER_FRACTION_MOE  = 4, // everything but sparse MoE weights23};24 25class common_params_fit_exception : public std::runtime_error {26    using std::runtime_error::runtime_error;27};28 29static std::vector<llama_device_memory_data> common_get_device_memory_data_impl(30        const char * path_model,31        const llama_model_params * mparams,32        const llama_context_params * cparams,33        std::vector<ggml_backend_dev_t> & devs,34        uint32_t & hp_ngl,35        uint32_t & hp_n_ctx_train,36        uint32_t & hp_n_expert,37        ggml_log_level log_level) {38    struct user_data_t {39        struct {40            ggml_log_callback callback;41            void * user_data;42        } original_logger;43        ggml_log_level min_level; // prints below this log level go to debug log44    };45    user_data_t ud;46    llama_log_get(&ud.original_logger.callback, &ud.original_logger.user_data);47    ud.min_level = log_level;48 49    llama_log_set([](ggml_log_level level, const char * text, void * user_data) {50        const user_data_t * ud = (const user_data_t *) user_data;51        const ggml_log_level level_eff = level >= ud->min_level ? level : GGML_LOG_LEVEL_DEBUG;52        ud->original_logger.callback(level_eff, text, ud->original_logger.user_data);53    }, &ud);54 55    llama_model_params mparams_copy = *mparams;56    mparams_copy.no_alloc  = true;57    mparams_copy.load_mode = LLAMA_LOAD_MODE_NONE;58 59    llama_model * model = llama_model_load_from_file(path_model, mparams_copy);60    if (model == nullptr) {61        llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);62        throw std::runtime_error("failed to load model");63    }64 65    llama_context * ctx = llama_init_from_model(model, *cparams);66    if (ctx == nullptr) {67        llama_model_free(model);68        llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);69        throw std::runtime_error("failed to create llama_context from model");70    }71 72    const size_t nd = llama_model_n_devices(model);73    std::vector<llama_device_memory_data> ret(nd + 1);74 75    llama_memory_breakdown memory_breakdown = llama_get_memory_breakdown(ctx);76 77    for (const auto & [buft, mb] : memory_breakdown) {78        if (ggml_backend_buft_is_host(buft)) {79            ret.back().mb.model   += mb.model;80            ret.back().mb.context += mb.context;81            ret.back().mb.compute += mb.compute;82            continue;83        }84 85        ggml_backend_dev_t dev = ggml_backend_buft_get_device(buft);86        if (!dev) {87            continue;88        }89        for (size_t i = 0; i < nd; i++) {90            if (dev == llama_model_get_device(model, i)) {91                ret[i].mb.model   += mb.model;92                ret[i].mb.context += mb.context;93                ret[i].mb.compute += mb.compute;94                break;95            }96        }97    }98 99    {100        ggml_backend_dev_t cpu_dev = ggml_backend_dev_by_type(GGML_BACKEND_DEVICE_TYPE_CPU);101        if (cpu_dev == nullptr) {102            throw std::runtime_error("no CPU backend found");103        }104        size_t free;105        size_t total;106        ggml_backend_dev_memory(cpu_dev, &free, &total);107        ret.back().free  = free;108        ret.back().total = total;109    }110    for (size_t i = 0; i < nd; i++) {111        ggml_backend_dev_t dev = llama_model_get_device(model, i);112 113        size_t free;114        size_t total;115        ggml_backend_dev_memory(dev, &free, &total);116 117        // Some non-GPU accelerator backends, such as BLAS, report 0/0 and rely on118        // the host-memory fallback. For GPU-like backends, keep 0/0 so --fit does119        // not assign anything to a device with an unknown memory budget.120        if (free == 0 && total == 0) {121            const enum ggml_backend_dev_type type = ggml_backend_dev_type(dev);122            if (type == GGML_BACKEND_DEVICE_TYPE_GPU || type == GGML_BACKEND_DEVICE_TYPE_IGPU) {123                LOG_WRN("%s: device %s did not report memory; --fit will not use it\n",124                        __func__, ggml_backend_dev_name(dev));125            } else {126                free  = ret.back().free;127                total = ret.back().total;128            }129        }130        ret[i].free  = free;131        ret[i].total = total;132    }133 134    devs.clear();135    for (int i = 0; i < llama_model_n_devices(model); i++) {136        devs.push_back(llama_model_get_device(model, i));137    }138 139    hp_ngl         = llama_model_n_layer(model);140    if (mparams->load_mtp) {141        hp_ngl    += llama_model_n_layer_nextn(model);142    }143    hp_n_ctx_train = llama_model_n_ctx_train(model);144    hp_n_expert    = llama_model_n_expert(model);145 146    common_memory_breakdown_print(ctx);147 148    llama_free(ctx);149    llama_model_free(model);150    llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);151 152    return ret;153}154 155common_device_memory_data_vec common_get_device_memory_data(156        const char * path_model,157        const llama_model_params * mparams,158        const llama_context_params * cparams,159        std::vector<ggml_backend_dev_t> & devs,160        uint32_t & hp_ngl,161        uint32_t & hp_n_ctx_train,162        uint32_t & hp_n_expert,163        ggml_log_level log_level) {164    std::vector<llama_device_memory_data> impl = common_get_device_memory_data_impl(165            path_model, mparams, cparams, devs, hp_ngl, hp_n_ctx_train, hp_n_expert, log_level);166 167    common_device_memory_data_vec ret(impl.size());168    for (size_t i = 0; i < impl.size(); i++) {169        ret[i].total   = impl[i].total;170        ret[i].free    = impl[i].free;171        ret[i].model   = impl[i].mb.model;172        ret[i].context = impl[i].mb.context;173        ret[i].compute = impl[i].mb.compute;174    }175    return ret;176}177 178static void common_params_fit_impl(179        const char * path_model, struct llama_model_params * mparams, struct llama_context_params * cparams,180        float * tensor_split, struct llama_model_tensor_buft_override * tensor_buft_overrides,181        size_t * margins_s, uint32_t n_ctx_min, enum ggml_log_level log_level) {182    if (mparams->split_mode == LLAMA_SPLIT_MODE_TENSOR) {183        throw common_params_fit_exception("llama_params_fit is not implemented for SPLIT_MODE_TENSOR, abort");184    }185    constexpr int64_t MiB = 1024*1024;186    typedef std::vector<llama_device_memory_data> dmds_t;187    const llama_model_params default_mparams = llama_model_default_params();188 189    std::vector<ggml_backend_dev_t> devs;190    uint32_t hp_ngl = 0; // hparams.n_gpu_layers191    uint32_t hp_nct = 0; // hparams.n_ctx_train192    uint32_t hp_nex = 0; // hparams.n_expert193 194    // step 1: get data for default parameters and check whether any changes are necessary in the first place195 196    LOG_TRC("%s: getting device memory data for initial parameters:\n", __func__);197    const dmds_t dmds_full = common_get_device_memory_data_impl(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);198    const size_t nd = devs.size(); // number of devices199 200    std::vector<int64_t> margins; // this function uses int64_t rather than size_t for memory sizes to more conveniently handle deficits201    margins.reserve(nd);202    if (nd == 0) {203        margins.push_back(margins_s[0]);204    } else {205        for (size_t id = 0; id < nd; id++) {206            margins.push_back(margins_s[id]);207        }208    }209 210    std::vector<std::string> dev_names;211    {212        dev_names.reserve(nd);213        size_t max_length = 0;214        for (const auto & dev : devs) {215            std::string name = ggml_backend_dev_name(dev);216            name += " (";217            name += ggml_backend_dev_description(dev);218            name += ")";219            dev_names.push_back(name);220            max_length = std::max(max_length, name.length());221        }222        for (std::string & dn : dev_names) {223            dn.insert(dn.end(), max_length - dn.length(), ' ');224        }225    }226 227    int64_t sum_free            = 0;228    int64_t sum_projected_free  = 0;229    int64_t sum_projected_used  = 0;230    int64_t sum_projected_model = 0;231    std::vector<int64_t> projected_free_per_device;232    projected_free_per_device.reserve(nd);233 234    if (nd == 0) {235        sum_projected_used = dmds_full.back().mb.total();236        sum_free           = dmds_full.back().total;237        sum_projected_free = sum_free - sum_projected_used;238        LOG_TRC("%s: projected to use %" PRId64 " MiB of host memory vs. %" PRId64 " MiB of total host memory\n",239            __func__, sum_projected_used/MiB, sum_free/MiB);240        if (sum_projected_free >= margins[0]) {241            LOG_TRC("%s: will leave %" PRId64 " >= %" PRId64 " MiB of system memory, no changes needed\n",242                __func__, sum_projected_free/MiB, margins[0]/MiB);243            return;244        }245    } else {246        if (nd > 1) {247            LOG_TRC("%s: projected memory use with initial parameters [MiB]:\n", __func__);248        }249        for (size_t id = 0; id < nd; id++) {250            const llama_device_memory_data & dmd = dmds_full[id];251 252            const int64_t projected_used = dmd.mb.total();253            const int64_t projected_free = dmd.free - projected_used;254            projected_free_per_device.push_back(projected_free);255 256            sum_free            += dmd.free;257            sum_projected_used  += projected_used;258            sum_projected_free  += projected_free;259            sum_projected_model += dmd.mb.model;260 261            if (nd > 1) {262                LOG_TRC("%s:   - %s: %6" PRId64 " total, %6" PRId64 " used, %6" PRId64 " free vs. target of %6" PRId64 "\n",263                    __func__, dev_names[id].c_str(), dmd.total/MiB, projected_used/MiB, projected_free/MiB, margins[id]/MiB);264            }265        }266        assert(sum_free >= 0 && sum_projected_used >= 0);267        LOG_TRC("%s: projected to use %" PRId64 " MiB of device memory vs. %" PRId64 " MiB of free device memory\n",268            __func__, sum_projected_used/MiB, sum_free/MiB);269        if (nd == 1) {270            if (projected_free_per_device[0] >= margins[0]) {271                LOG_TRC("%s: will leave %" PRId64 " >= %" PRId64 " MiB of free device memory, no changes needed\n",272                    __func__, projected_free_per_device[0]/MiB, margins[0]/MiB);273                return;274            }275        } else {276            bool changes_needed = false;277            for (size_t id = 0; id < nd; id++) {278                if (projected_free_per_device[id] < margins[id]) {279                    changes_needed = true;280                    break;281                }282            }283            if (!changes_needed) {284                LOG_TRC("%s: targets for free memory can be met on all devices, no changes needed\n", __func__);285                return;286            }287        }288    }289 290    // step 2: try reducing memory use by reducing the context size291 292    {293        int64_t global_surplus = sum_projected_free;294        if (nd == 0) {295            global_surplus -= margins[0];296        } else {297            for (size_t id = 0; id < nd; id++) {298                global_surplus -= margins[id];299            }300        }301        if (global_surplus < 0) {302            if (nd <= 1) {303                LOG_TRC("%s: cannot meet free memory target of %" PRId64 " MiB, need to reduce device memory by %" PRId64 " MiB\n",304                    __func__, margins[0]/MiB, -global_surplus/MiB);305            } else {306                LOG_TRC(307                    "%s: cannot meet free memory targets on all devices, need to use %" PRId64 " MiB less in total\n",308                    __func__, -global_surplus/MiB);309            }310            if (cparams->n_ctx == 0) {311                if (hp_nct > n_ctx_min) {312                    int64_t sum_used_target = sum_free;313                    if (nd == 0) {314                        sum_used_target -= margins[0];315                    } else {316                        for (size_t id = 0; id < nd; id++) {317                            sum_used_target -= margins[id];318                        }319                    }320                    if (nd > 1) {321                        // for multiple devices we need to be more conservative in terms of how much context we think can fit:322                        //   - for dense models only whole layers can be assigned to devices323                        //   - for MoE models only whole tensors can be assigned to devices, which we estimate to be <= 1/3 of a layer324                        //   - on average we expect a waste of 0.5 layers/tensors per device325                        //   - use slightly more than the expected average for nd devices to be safe326                        const int64_t model_per_layer = sum_projected_model / std::min(uint32_t(mparams->n_gpu_layers), hp_ngl);327                        sum_used_target -= (nd + 1) * model_per_layer / (hp_nex == 0 ? 2 : 6);328                    }329 330                    int64_t sum_projected_used_min_ctx = 0;331                    cparams->n_ctx = n_ctx_min;332                    const dmds_t dmds_min_ctx = common_get_device_memory_data_impl(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);333                    if (nd == 0) {334                        sum_projected_used_min_ctx = dmds_min_ctx.back().mb.total();335                    } else {336                        for (size_t id = 0; id < nd; id++) {337                            sum_projected_used_min_ctx += dmds_min_ctx[id].mb.total();338                        }339                    }340                    if (sum_used_target > sum_projected_used_min_ctx) {341                        // linear interpolation between minimum and maximum context size:342                        cparams->n_ctx += (hp_nct - n_ctx_min) * (sum_used_target - sum_projected_used_min_ctx)343                            / (sum_projected_used - sum_projected_used_min_ctx);344                        cparams->n_ctx = std::max(cparams->n_ctx - cparams->n_ctx % 256, n_ctx_min); // round down context for CUDA backend345 346                        const int64_t bytes_per_ctx = (sum_projected_used - sum_projected_used_min_ctx) / (hp_nct - n_ctx_min);347                        const int64_t memory_reduction = (hp_nct - cparams->n_ctx) * bytes_per_ctx;348                        LOG_TRC("%s: context size reduced from %" PRIu32 " to %" PRIu32 " -> need %" PRId64 " MiB less memory in total\n",349                            __func__, hp_nct, cparams->n_ctx, memory_reduction/MiB);350                        if (nd <= 1) {351                            LOG_TRC("%s: entire model can be fit by reducing context\n", __func__);352                            return;353                        }354                        LOG_TRC("%s: entire model should be fit across devices by reducing context\n", __func__);355                    } else {356                        const int64_t memory_reduction = sum_projected_used - sum_projected_used_min_ctx;357                        LOG_TRC("%s: context size reduced from %" PRIu32 " to %" PRIu32 " -> need %" PRId64 " MiB less memory in total\n",358                            __func__, hp_nct, cparams->n_ctx, memory_reduction/MiB);359                    }360                } else {361                    if (n_ctx_min == UINT32_MAX) {362                        LOG_TRC("%s: user has requested full context size of %" PRIu32 " -> no change\n", __func__, hp_nct);363                    } else {364                        LOG_TRC("%s: default model context size is %" PRIu32 " which is <= the min. context size of %" PRIu32 " -> no change\n",365                            __func__, hp_nct, n_ctx_min);366                    }367                }368            } else {369                LOG_TRC("%s: context size set by user to %" PRIu32 " -> no change\n", __func__, cparams->n_ctx);370            }371        }372    }373    if (nd == 0) {374        throw common_params_fit_exception("was unable to fit model into system memory by reducing context, abort");375    }376 377    if (mparams->n_gpu_layers != default_mparams.n_gpu_layers) {378        throw common_params_fit_exception("n_gpu_layers already set by user to " + std::to_string(mparams->n_gpu_layers) + ", abort");379    }380    if (nd > 1) {381        if (!tensor_split) {382            throw common_params_fit_exception("did not provide a buffer to write the tensor_split to, abort");383        }384        if (mparams->tensor_split) {385            for (size_t id = 0; id < nd; id++) {386                if (mparams->tensor_split[id] != 0.0f) {387                    throw common_params_fit_exception("model_params::tensor_split already set by user, abort");388                }389            }390        }391        if (mparams->split_mode == LLAMA_SPLIT_MODE_ROW) {392            throw common_params_fit_exception("changing weight allocation for LLAMA_SPLIT_MODE_ROW not implemented, abort");393        }394    }395    if (!tensor_buft_overrides) {396        throw common_params_fit_exception("did not provide buffer to set tensor_buft_overrides, abort");397    }398    if (mparams->tensor_buft_overrides && (mparams->tensor_buft_overrides->pattern || mparams->tensor_buft_overrides->buft)) {399        throw common_params_fit_exception("model_params::tensor_buft_overrides already set by user, abort");400    }401 402    // step 3: iteratively fill the back to front with "dense" layers403    //   - for a dense model simply fill full layers, giving each device a contiguous slice of the model404    //   - for a MoE model, same as dense model but with all MoE tensors in system memory405 406    // utility function that returns a static C string matching the tensors for a specific layer index and layer fraction:407    auto get_overflow_pattern = [&](const size_t il, const common_layer_fraction_t lf) -> const char * {408        constexpr size_t n_strings = 1000;409        if (il >= n_strings) {410            throw std::runtime_error("at most " + std::to_string(n_strings) + " model layers are supported");411        }412        switch (lf) {413            case LAYER_FRACTION_ATTN: {414                static std::array<std::string, n_strings> patterns;415                if (patterns[il].empty()) {416                    patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(gate|up|gate_up|down).*";417                }418                return patterns[il].c_str();419            }420            case LAYER_FRACTION_UP: {421                static std::array<std::string, n_strings> patterns;422                if (patterns[il].empty()) {423                    patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(gate|gate_up|down).*";424                }425                return patterns[il].c_str();426            }427            case LAYER_FRACTION_GATE: {428                static std::array<std::string, n_strings> patterns;429                if (patterns[il].empty()) {430                    patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_down.*";431                }432                return patterns[il].c_str();433            }434            case LAYER_FRACTION_MOE: {435                static std::array<std::string, n_strings> patterns;436                if (patterns[il].empty()) {437                    patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(up|down|gate_up|gate)_(ch|)exps";438                }439                return patterns[il].c_str();440            }441            default:442                GGML_ABORT("fatal error");443        }444    };445 446    struct ngl_t {447        uint32_t n_layer = 0; // number of total layers448        uint32_t n_part  = 0; // number of partial layers, <= n_layer449 450        // for the first partial layer varying parts can overflow, all further layers use LAYER_FRACTION_MOE:451        common_layer_fraction_t overflow_type = LAYER_FRACTION_MOE;452 453        uint32_t n_full() const {454            assert(n_layer >= n_part);455            return n_layer - n_part;456        }457    };458 459    const size_t ntbo = llama_max_tensor_buft_overrides();460 461    // utility function to set n_gpu_layers and tensor_split462    auto set_ngl_tensor_split_tbo = [&](463            const std::vector<ngl_t> & ngl_per_device,464            const std::vector<ggml_backend_buffer_type_t> & overflow_bufts,465            llama_model_params & mparams) {466        mparams.n_gpu_layers = 0;467        for (size_t id = 0; id < nd; id++) {468            mparams.n_gpu_layers += ngl_per_device[id].n_layer;469            if (nd > 1) {470                tensor_split[id] = ngl_per_device[id].n_layer;471            }472        }473        assert(uint32_t(mparams.n_gpu_layers) <= hp_ngl + 1);474        uint32_t il0 = hp_ngl + 1 - mparams.n_gpu_layers; // start index for tensor buft overrides475 476        mparams.tensor_split = tensor_split;477 478        size_t itbo = 0;479        for (size_t id = 0; id < nd; id++) {480            il0 += ngl_per_device[id].n_full();481            for (uint32_t il = il0; il < il0 + ngl_per_device[id].n_part; il++) {482                if (itbo + 1 >= ntbo) {483                    tensor_buft_overrides[itbo].pattern = nullptr;484                    tensor_buft_overrides[itbo].buft    = nullptr;485                    itbo++;486                    mparams.tensor_buft_overrides = tensor_buft_overrides;487                    throw common_params_fit_exception("llama_max_tensor_buft_overrides() == "488                        + std::to_string(ntbo) + " is insufficient for model");489                }490                tensor_buft_overrides[itbo].pattern = get_overflow_pattern(il, il == il0 ? ngl_per_device[id].overflow_type : LAYER_FRACTION_MOE);491                tensor_buft_overrides[itbo].buft = il == il0 ? overflow_bufts[id] : ggml_backend_cpu_buffer_type();492                itbo++;493            }494            il0 += ngl_per_device[id].n_part;495        }496        tensor_buft_overrides[itbo].pattern = nullptr;497        tensor_buft_overrides[itbo].buft    = nullptr;498        itbo++;499        mparams.tensor_buft_overrides = tensor_buft_overrides;500    };501 502    // utility function that returns the memory use per device for given numbers of layers per device503    auto get_memory_for_layers = [&](504            const char * func_name,505            const std::vector<ngl_t> & ngl_per_device,506            const std::vector<ggml_backend_buffer_type_t> & overflow_bufts) -> std::vector<int64_t> {507        llama_model_params mparams_copy = *mparams;508        set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, mparams_copy);509 510        const dmds_t dmd_nl = common_get_device_memory_data_impl(511            path_model, &mparams_copy, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);512 513        LOG_TRC("%s: memory for test allocation by device:\n", func_name);514        for (size_t id = 0; id < nd; id++) {515            const ngl_t & n = ngl_per_device[id];516            LOG_TRC(517                "%s: id=%zu, n_layer=%2" PRIu32 ", n_part=%2" PRIu32 ", overflow_type=%d, mem=%6" PRId64 " MiB\n",518                func_name, id, n.n_layer, n.n_part, int(n.overflow_type), dmd_nl[id].mb.total()/MiB);519        }520 521        std::vector<int64_t> ret;522        ret.reserve(nd);523        for (size_t id = 0; id < nd; id++) {524            ret.push_back(dmd_nl[id].mb.total());525        }526        return ret;527    };528 529    int64_t global_surplus_cpu_moe = 0;530    if (hp_nex > 0) {531        const static std::string pattern_moe_all = "blk\\.\\d+\\.ffn_(up|down|gate_up|gate)_(ch|)exps"; // matches all MoE tensors532        ggml_backend_buffer_type_t cpu_buft = ggml_backend_cpu_buffer_type();533        tensor_buft_overrides[0] = {pattern_moe_all.c_str(), cpu_buft};534        tensor_buft_overrides[1] = {nullptr, nullptr};535        mparams->tensor_buft_overrides = tensor_buft_overrides;536 537        LOG_TRC("%s: getting device memory data with all MoE tensors moved to system memory:\n", __func__);538        const dmds_t dmds_cpu_moe = common_get_device_memory_data_impl(539            path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);540 541        for (size_t id = 0; id < nd; id++) {542            global_surplus_cpu_moe += dmds_cpu_moe[id].free;543            global_surplus_cpu_moe -= int64_t(dmds_cpu_moe[id].mb.total()) + margins[id];544        }545 546        if (global_surplus_cpu_moe > 0) {547            LOG_TRC("%s: with only dense weights in device memory there is a total surplus of %" PRId64 " MiB\n",548                __func__, global_surplus_cpu_moe/MiB);549        } else {550            LOG_TRC("%s: with only dense weights in device memory there is still a total deficit of %" PRId64 " MiB\n",551                __func__, -global_surplus_cpu_moe/MiB);552        }553 554        // reset555        tensor_buft_overrides[0] = {nullptr, nullptr};556        mparams->tensor_buft_overrides = tensor_buft_overrides;557    }558 559    std::vector<int64_t> targets; // maximum acceptable memory use per device560    targets.reserve(nd);561    for (size_t id = 0; id < nd; id++) {562        targets.push_back(dmds_full[id].free - margins[id]);563        LOG_TRC("%s: id=%zu, target=%" PRId64 " MiB\n", __func__, id, targets[id]/MiB);564    }565 566    std::vector<ggml_backend_buffer_type_t> overflow_bufts; // which bufts the first partial layer of a device overflows to:567    overflow_bufts.reserve(nd);568    for (size_t id = 0; id < nd; id++) {569        overflow_bufts.push_back(ggml_backend_cpu_buffer_type());570    }571 572    std::vector<ngl_t> ngl_per_device(nd);573    std::vector<int64_t> mem = get_memory_for_layers(__func__, ngl_per_device, overflow_bufts);574 575    // optimize the number of layers per device using the method of false position:576    //   - ngl_per_device has 0 layers for each device, lower bound577    //   - try a "high" configuration where a device is given all unassigned layers578    //   - interpolate the memory use / layer between low and high linearly to get a guess where it meets our target579    //   - check memory use of our guess, replace either the low or high bound580    //   - once we only have a difference of a single layer, stop and return the lower bound that just barely still fits581    //   - the last device has the output layer, which cannot be a partial layer582    if (hp_nex == 0) {583        LOG_TRC("%s: filling dense layers back-to-front:\n", __func__);584    } else {585        LOG_TRC("%s: filling dense-only layers back-to-front:\n", __func__);586    }587    for (int id = nd - 1; id >= 0; id--) {588        uint32_t n_unassigned = hp_ngl + 1;589        for (size_t jd = id + 1; jd < nd; ++jd) {590            assert(n_unassigned >= ngl_per_device[jd].n_layer);591            n_unassigned -= ngl_per_device[jd].n_layer;592        }593 594        std::vector<ngl_t> ngl_per_device_high = ngl_per_device;595        ngl_per_device_high[id].n_layer = n_unassigned;596        if (hp_nex > 0) {597            ngl_per_device_high[id].n_part = size_t(id) < nd - 1 ? ngl_per_device_high[id].n_layer : ngl_per_device_high[id].n_layer - 1;598        }599        if (ngl_per_device_high[id].n_layer > 0) {600            std::vector<int64_t> mem_high = get_memory_for_layers(__func__, ngl_per_device_high, overflow_bufts);601            if (mem_high[id] > targets[id]) {602                assert(ngl_per_device_high[id].n_layer > ngl_per_device[id].n_layer);603                uint32_t delta = ngl_per_device_high[id].n_layer - ngl_per_device[id].n_layer;604                LOG_TRC("%s: start filling device %" PRIu32 ", delta=%" PRIu32 "\n", __func__, id, delta);605                while (delta > 1) {606                    uint32_t step_size = int64_t(delta) * (targets[id] - mem[id]) / (mem_high[id] - mem[id]);607                    step_size = std::max(step_size, uint32_t(1));608                    step_size = std::min(step_size, delta - 1);609 610                    std::vector<ngl_t> ngl_per_device_test = ngl_per_device;611                    ngl_per_device_test[id].n_layer += step_size;612                    if (hp_nex) {613                        ngl_per_device_test[id].n_part += size_t(id) == nd - 1 && ngl_per_device_test[id].n_part == 0 ?614                            step_size - 1 : step_size; // the first layer is the output layer which must always be full615                    }616                    const std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts);617 618                    if (mem_test[id] <= targets[id]) {619                        ngl_per_device = ngl_per_device_test;620                        mem            = mem_test;621                        LOG_TRC("%s: set ngl_per_device[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device[id].n_layer);622                    } else {623                        ngl_per_device_high = ngl_per_device_test;624                        mem_high            = mem_test;625                        LOG_TRC("%s: set ngl_per_device_high[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device_high[id].n_layer);626                    }627                    delta = ngl_per_device_high[id].n_layer - ngl_per_device[id].n_layer;628                }629            } else {630                assert(ngl_per_device_high[id].n_layer == n_unassigned);631                ngl_per_device = ngl_per_device_high;632                mem            = mem_high;633                LOG_TRC("%s: set ngl_per_device[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device[id].n_layer);634            }635        }636 637        const int64_t projected_margin = dmds_full[id].free - mem[id];638        LOG_TRC(639            "%s:   - %s: %2" PRIu32 " layers, %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",640            __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, mem[id]/MiB, projected_margin/MiB);641    }642    if (hp_nex == 0 || global_surplus_cpu_moe <= 0) {643        set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, *mparams);644        return;645    }646 647    // step 4: for a MoE model where all dense tensors fit,648    //     convert the dense-only layers in the back to full layers in the front until all devices are full649    // essentially the same procedure as for the dense-only layers except front-to-back650    // also, try fitting at least part of one more layer to reduce waste for "small" GPUs with e.g. 24 GiB VRAM651 652    size_t id_dense_start = nd;653    for (int id = nd - 1; id >= 0; id--) {654        if (ngl_per_device[id].n_layer > 0) {655            id_dense_start = id;656            continue;657        }658        break;659    }660    assert(id_dense_start < nd);661 662    LOG_TRC("%s: converting dense-only layers to full layers and filling them front-to-back with overflow to next device/system memory:\n", __func__);663    for (size_t id = 0; id <= id_dense_start && id_dense_start < nd; id++) {664        std::vector<ngl_t> ngl_per_device_high = ngl_per_device;665        for (size_t jd = id_dense_start; jd < nd; jd++) {666            const uint32_t n_layer_move = jd < nd - 1 ? ngl_per_device_high[jd].n_layer : ngl_per_device_high[jd].n_layer - 1;667            ngl_per_device_high[id].n_layer += n_layer_move;668            ngl_per_device_high[jd].n_layer -= n_layer_move;669            ngl_per_device_high[jd].n_part = 0;670        }671        size_t id_dense_start_high = nd - 1;672        std::vector<int64_t> mem_high = get_memory_for_layers(__func__, ngl_per_device_high, overflow_bufts);673 674        if (mem_high[id] > targets[id]) {675            assert(ngl_per_device_high[id].n_full() >= ngl_per_device[id].n_full());676            uint32_t delta = ngl_per_device_high[id].n_full() - ngl_per_device[id].n_full();677            while (delta > 1) {678                uint32_t step_size = int64_t(delta) * (targets[id] - mem[id]) / (mem_high[id] - mem[id]);679                step_size = std::max(step_size, uint32_t(1));680                step_size = std::min(step_size, delta - 1);681 682                std::vector<ngl_t> ngl_per_device_test = ngl_per_device;683                size_t id_dense_start_test = id_dense_start;684                uint32_t n_converted_test = 0;685                for (;id_dense_start_test < nd; id_dense_start_test++) {686                    const uint32_t n_convert_jd = std::min(step_size - n_converted_test, ngl_per_device_test[id_dense_start_test].n_part);687                    ngl_per_device_test[id_dense_start_test].n_layer -= n_convert_jd;688                    ngl_per_device_test[id_dense_start_test].n_part -= n_convert_jd;689                    ngl_per_device_test[id].n_layer += n_convert_jd;690                    n_converted_test += n_convert_jd;691 692                    if (ngl_per_device_test[id_dense_start_test].n_part > 0) {693                        break;694                    }695                }696                const std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts);697 698                if (mem_test[id] <= targets[id]) {699                    ngl_per_device = ngl_per_device_test;700                    mem            = mem_test;701                    id_dense_start = id_dense_start_test;702                    LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start=%zu\n",703                        __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);704                } else {705                    ngl_per_device_high = ngl_per_device_test;706                    mem_high            = mem_test;707                    id_dense_start_high = id_dense_start_test;708                    LOG_TRC("%s: set ngl_per_device_high[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start_high=%zu\n",709                        __func__, id, ngl_per_device_high[id].n_layer, ngl_per_device_high[id].n_part, id_dense_start_high);710                }711                assert(ngl_per_device_high[id].n_full() >= ngl_per_device[id].n_full());712                delta = ngl_per_device_high[id].n_full() - ngl_per_device[id].n_full();713            }714        } else {715            ngl_per_device = ngl_per_device_high;716            mem            = mem_high;717            id_dense_start = id_dense_start_high;718            LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start=%zu\n",719                __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);720        }721 722        // try to fit at least part of one more layer723        if (ngl_per_device[id_dense_start].n_layer > (id < nd - 1 ? 0 : 1)) {724            std::vector<ngl_t> ngl_per_device_test = ngl_per_device;725            size_t id_dense_start_test = id_dense_start;726            ngl_per_device_test[id_dense_start_test].n_layer--;727            ngl_per_device_test[id_dense_start_test].n_part--;728            ngl_per_device_test[id].n_layer++;729            ngl_per_device_test[id].n_part++;730            if (ngl_per_device_test[id_dense_start_test].n_part == 0) {731                id_dense_start_test++;732            }733            ngl_per_device_test[id].overflow_type = LAYER_FRACTION_UP;734            std::vector<ggml_backend_buffer_type_t> overflow_bufts_test = overflow_bufts;735            if (id < nd - 1) {736                overflow_bufts_test[id] = ggml_backend_dev_buffer_type(devs[id + 1]);737            }738            LOG_TRC("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_UP\n", __func__);739            std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);740            if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {741                ngl_per_device = ngl_per_device_test;742                overflow_bufts = overflow_bufts_test;743                mem            = mem_test;744                id_dense_start = id_dense_start_test;745                LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", UP), id_dense_start=%zu\n",746                    __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);747 748                ngl_per_device_test[id].overflow_type = LAYER_FRACTION_GATE;749                LOG_TRC("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_GATE\n", __func__);750                mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);751                if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {752                    ngl_per_device = ngl_per_device_test;753                    overflow_bufts = overflow_bufts_test;754                    mem            = mem_test;755                    id_dense_start = id_dense_start_test;756                    LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", GATE), id_dense_start=%zu\n",757                        __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);758                }759            } else {760                ngl_per_device_test[id].overflow_type = LAYER_FRACTION_ATTN;761                LOG_TRC("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_ATTN\n", __func__);762                mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);763                if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {764                    ngl_per_device = ngl_per_device_test;765                    overflow_bufts = overflow_bufts_test;766                    mem            = mem_test;767                    id_dense_start = id_dense_start_test;768                    LOG_TRC("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", ATTN), id_dense_start=%zu\n",769                        __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);770                }771            }772        }773 774        const int64_t projected_margin = dmds_full[id].free - mem[id];775        LOG_TRC(776            "%s:   - %s: %2" PRIu32 " layers (%2" PRIu32 " overflowing), %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",777            __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, ngl_per_device[id].n_part, mem[id]/MiB, projected_margin/MiB);778    }779 780    // print info for devices that were not changed during the conversion from dense only to full layers:781    for (size_t id = id_dense_start + 1; id < nd; id++) {782        const int64_t projected_margin = dmds_full[id].free - mem[id];783        LOG_TRC(784            "%s:   - %s: %2" PRIu32 " layers (%2" PRIu32 " overflowing), %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",785            __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, ngl_per_device[id].n_part, mem[id]/MiB, projected_margin/MiB);786    }787 788    set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, *mparams);789}790 791enum common_params_fit_status common_fit_params(792        const char * path_model,793        llama_model_params * mparams,794        llama_context_params * cparams,795        float * tensor_split,796        llama_model_tensor_buft_override * tensor_buft_overrides,797        size_t * margins,798        uint32_t n_ctx_min,799        ggml_log_level log_level) {800    const int64_t t0_us = llama_time_us();801    common_params_fit_status status = COMMON_PARAMS_FIT_STATUS_SUCCESS;802    try {803        common_params_fit_impl(path_model, mparams, cparams, tensor_split, tensor_buft_overrides, margins, n_ctx_min, log_level);804        LOG_TRC("%s: successfully fit params to free device memory\n", __func__);805    } catch (const common_params_fit_exception & e) {806        LOG_WRN("%s: failed to fit params to free device memory: %s\n", __func__, e.what());807        status = COMMON_PARAMS_FIT_STATUS_FAILURE;808    } catch (const std::runtime_error & e) {809        LOG_ERR("%s: encountered an error while trying to fit params to free device memory: %s\n", __func__, e.what());810        status = COMMON_PARAMS_FIT_STATUS_ERROR;811    }812    const int64_t t1_us = llama_time_us();813    LOG_TRC("%s: fitting params to free memory took %.2f seconds\n", __func__, (t1_us - t0_us) * 1e-6);814    return status;815}816 817void common_memory_breakdown_print(const struct llama_context * ctx) {818    //const auto & devices = ctx->get_model().devices;819    const auto * model = llama_get_model(ctx);820 821    std::vector<ggml_backend_dev_t> devices;822    for (int i = 0; i < llama_model_n_devices(model); i++) {823        devices.push_back(llama_model_get_device(model, i));824    }825 826    llama_memory_breakdown memory_breakdown = llama_get_memory_breakdown(ctx);827 828    std::vector<std::array<std::string, 9>> table_data;829    table_data.reserve(devices.size());830    const std::string template_header = "%s: | %s | %s   %s    %s   %s   %s   %s    %s |\n";831    const std::string template_gpu    = "%s: | %s | %s = %s + (%s = %s + %s + %s) + %s |\n";832    const std::string template_other  = "%s: | %s | %s   %s    %s = %s + %s + %s    %s |\n";833 834    table_data.push_back({template_header, "memory breakdown [MiB]", "total", "free", "self", "model", "context", "compute", "unaccounted"});835 836    constexpr size_t MiB = 1024 * 1024;837    const std::vector<std::string> desc_prefixes_strip = {"NVIDIA ", "GeForce ", "Tesla ", "AMD ", "Radeon ", "Instinct "};838 839    // track seen buffer types to avoid double counting:840    std::set<ggml_backend_buffer_type_t> seen_buffer_types;841 842    // accumulative memory breakdown for each device and for host:843    std::vector<llama_memory_breakdown_data> mb_dev(devices.size());844    llama_memory_breakdown_data              mb_host;845 846    for (const auto & buft_mb : memory_breakdown) {847        ggml_backend_buffer_type_t          buft = buft_mb.first;848        const llama_memory_breakdown_data & mb   = buft_mb.second;849        if (ggml_backend_buft_is_host(buft)) {850            mb_host.model   += mb.model;851            mb_host.context += mb.context;852            mb_host.compute += mb.compute;853            seen_buffer_types.insert(buft);854            continue;855        }856        ggml_backend_dev_t dev = ggml_backend_buft_get_device(buft);857        if (dev) {858            int i_dev = -1;859            for (size_t i = 0; i < devices.size(); i++) {860                if (devices[i] == dev) {861                    i_dev = i;862                    break;863                }864            }865            if (i_dev != -1) {866                mb_dev[i_dev].model   += mb.model;867                mb_dev[i_dev].context += mb.context;868                mb_dev[i_dev].compute += mb.compute;869                seen_buffer_types.insert(buft);870                continue;871            }872        }873    }874 875    // print memory breakdown for each device:876    for (size_t i = 0; i < devices.size(); i++) {877        ggml_backend_dev_t dev = devices[i];878        llama_memory_breakdown_data mb = mb_dev[i];879 880        const std::string name = ggml_backend_dev_name(dev);881        std::string desc = ggml_backend_dev_description(dev);882        for (const std::string & prefix : desc_prefixes_strip) {883            if (desc.length() >= prefix.length() && desc.substr(0, prefix.length()) == prefix) {884                desc = desc.substr(prefix.length());885            }886        }887 888        size_t free, total;889        ggml_backend_dev_memory(dev, &free, &total);890 891        const size_t self = mb.model + mb.context + mb.compute;892        const int64_t unaccounted = static_cast<int64_t>(total) - static_cast<int64_t>(free) - static_cast<int64_t>(self);893 894        table_data.push_back({895            template_gpu,896            "  - " + name + " (" + desc + ")",897            std::to_string(total / MiB),898            std::to_string(free / MiB),899            std::to_string(self / MiB),900            std::to_string(mb.model / MiB),901            std::to_string(mb.context / MiB),902            std::to_string(mb.compute / MiB),903            std::to_string(unaccounted / static_cast<int64_t>(MiB))});904    }905 906    // print memory breakdown for host:907    {908        const size_t self = mb_host.model + mb_host.context + mb_host.compute;909        table_data.push_back({910            template_other,911            "  - Host",912            "", // total913            "", // free914            std::to_string(self / MiB),915            std::to_string(mb_host.model / MiB),916            std::to_string(mb_host.context / MiB),917            std::to_string(mb_host.compute / MiB),918            ""}); // unaccounted919    }920 921    // print memory breakdown for all remaining buffer types:922    for (const auto & buft_mb : memory_breakdown) {923        ggml_backend_buffer_type_t          buft = buft_mb.first;924        const llama_memory_breakdown_data & mb   = buft_mb.second;925        if (seen_buffer_types.count(buft) == 1) {926            continue;927        }928        const std::string name = ggml_backend_buft_name(buft);929        const size_t self = mb.model + mb.context + mb.compute;930        table_data.push_back({931            template_other,932            "  - " + name,933            "", // total934            "", // free935            std::to_string(self / MiB),936            std::to_string(mb.model / MiB),937            std::to_string(mb.context / MiB),938            std::to_string(mb.compute / MiB),939            ""}); // unaccounted940        seen_buffer_types.insert(buft);941    }942 943    for (size_t j = 1; j < table_data[0].size(); j++) {944        size_t max_len = 0;945        for (const auto & td : table_data) {946            max_len = std::max(max_len, td[j].length());947        }948        for (auto & td : table_data) {949            td[j].insert(j == 1 ? td[j].length() : 0, max_len - td[j].length(), ' ');950        }951    }952    for (const auto & td : table_data) {953        LOG_TRC(td[0].c_str(),954            __func__, td[1].c_str(), td[2].c_str(), td[3].c_str(), td[4].c_str(), td[5].c_str(),955            td[6].c_str(), td[7].c_str(), td[8].c_str());956    }957}958 959void common_fit_print(960        const char * path_model,961        llama_model_params * mparams,962        llama_context_params * cparams) {963    std::vector<ggml_backend_dev_t> devs;964    uint32_t hp_ngl = 0; // hparams.n_gpu_layers965    uint32_t hp_nct = 0; // hparams.n_ctx_train966    uint32_t hp_nex = 0; // hparams.n_expert967 968    auto dmd = common_get_device_memory_data_impl(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, GGML_LOG_LEVEL_ERROR);969    GGML_ASSERT(dmd.size() == devs.size() + 1);970 971    for (size_t id = 0; id < devs.size(); id++) {972        printf("%s ",  ggml_backend_dev_name(devs[id]));973        printf("%zu ", dmd[id].mb.model/1024/1024);974        printf("%zu ", dmd[id].mb.context/1024/1024);975        printf("%zu ", dmd[id].mb.compute/1024/1024);976        printf("\n");977    }978 979    printf("Host ");980    printf("%zu ", dmd.back().mb.model/1024/1024);981    printf("%zu ", dmd.back().mb.context/1024/1024);982    printf("%zu ", dmd.back().mb.compute/1024/1024);983    printf("\n");984}985