Team Ai
Datasetpublic

echodict/llama.cpp

version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes479downloads
fit.cpp952 linesDownload Raw Back to common
1#include "fit.h"2 3#include "log.h"4 5#include "../src/llama-ext.h"6 7#include <array>8#include <cassert>9#include <stdexcept>10#include <cinttypes>11#include <set>12#include <string>13#include <vector>14 15// this enum is only used in llama_params_fit_impl but needs to be defined outside of it to fix a Windows compilation issue16// enum to identify part of a layer for distributing its tensors:17enum common_layer_fraction_t {18    LAYER_FRACTION_NONE = 0, // nothing19    LAYER_FRACTION_ATTN = 1, // attention20    LAYER_FRACTION_UP   = 2, // attention + up21    LAYER_FRACTION_GATE = 3, // attention + up + gate22    LAYER_FRACTION_MOE  = 4, // everything but sparse MoE weights23};24 25class common_params_fit_exception : public std::runtime_error {26    using std::runtime_error::runtime_error;27};28 29static std::vector<llama_device_memory_data> common_get_device_memory_data(30        const char * path_model,31        const llama_model_params * mparams,32        const llama_context_params * cparams,33        std::vector<ggml_backend_dev_t> & devs,34        uint32_t & hp_ngl,35        uint32_t & hp_n_ctx_train,36        uint32_t & hp_n_expert,37        ggml_log_level log_level) {38    struct user_data_t {39        struct {40            ggml_log_callback callback;41            void * user_data;42        } original_logger;43        ggml_log_level min_level; // prints below this log level go to debug log44    };45    user_data_t ud;46    llama_log_get(&ud.original_logger.callback, &ud.original_logger.user_data);47    ud.min_level = log_level;48 49    llama_log_set([](ggml_log_level level, const char * text, void * user_data) {50        const user_data_t * ud = (const user_data_t *) user_data;51        const ggml_log_level level_eff = level >= ud->min_level ? level : GGML_LOG_LEVEL_DEBUG;52        ud->original_logger.callback(level_eff, text, ud->original_logger.user_data);53    }, &ud);54 55    llama_model_params mparams_copy = *mparams;56    mparams_copy.no_alloc  = true;57    mparams_copy.use_mmap  = false;58    mparams_copy.use_mlock = false;59 60    llama_model * model = llama_model_load_from_file(path_model, mparams_copy);61    if (model == nullptr) {62        llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);63        throw std::runtime_error("failed to load model");64    }65 66    llama_context * ctx = llama_init_from_model(model, *cparams);67    if (ctx == nullptr) {68        llama_model_free(model);69        llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);70        throw std::runtime_error("failed to create llama_context from model");71    }72 73    const size_t nd = llama_model_n_devices(model);74    std::vector<llama_device_memory_data> ret(nd + 1);75 76    llama_memory_breakdown memory_breakdown = llama_get_memory_breakdown(ctx);77 78    for (const auto & [buft, mb] : memory_breakdown) {79        if (ggml_backend_buft_is_host(buft)) {80            ret.back().mb.model   += mb.model;81            ret.back().mb.context += mb.context;82            ret.back().mb.compute += mb.compute;83            continue;84        }85 86        ggml_backend_dev_t dev = ggml_backend_buft_get_device(buft);87        if (!dev) {88            continue;89        }90        for (size_t i = 0; i < nd; i++) {91            if (dev == llama_model_get_device(model, i)) {92                ret[i].mb.model   += mb.model;93                ret[i].mb.context += mb.context;94                ret[i].mb.compute += mb.compute;95                break;96            }97        }98    }99 100    {101        ggml_backend_dev_t cpu_dev = ggml_backend_dev_by_type(GGML_BACKEND_DEVICE_TYPE_CPU);102        if (cpu_dev == nullptr) {103            throw std::runtime_error("no CPU backend found");104        }105        size_t free;106        size_t total;107        ggml_backend_dev_memory(cpu_dev, &free, &total);108        ret.back().free  = free;109        ret.back().total = total;110    }111    for (size_t i = 0; i < nd; i++) {112        size_t free;113        size_t total;114        ggml_backend_dev_memory(llama_model_get_device(model, i), &free, &total);115 116        // devices can return 0 bytes for free and total memory if they do not117        // have any to report. in this case, we will use the host memory as a fallback118        // fixes: https://github.com/ggml-org/llama.cpp/issues/18577119        if (free == 0 && total == 0) {120            free  = ret.back().free;121            total = ret.back().total;122        }123        ret[i].free  = free;124        ret[i].total = total;125    }126 127    devs.clear();128    for (int i = 0; i < llama_model_n_devices(model); i++) {129        devs.push_back(llama_model_get_device(model, i));130    }131 132    hp_ngl         = llama_model_n_layer(model);133    hp_n_ctx_train = llama_model_n_ctx_train(model);134    hp_n_expert    = llama_model_n_expert(model);135 136    common_memory_breakdown_print(ctx);137 138    llama_free(ctx);139    llama_model_free(model);140    llama_log_set(ud.original_logger.callback, ud.original_logger.user_data);141 142    return ret;143}144 145static void common_params_fit_impl(146        const char * path_model, struct llama_model_params * mparams, struct llama_context_params * cparams,147        float * tensor_split, struct llama_model_tensor_buft_override * tensor_buft_overrides,148        size_t * margins_s, uint32_t n_ctx_min, enum ggml_log_level log_level) {149    if (mparams->split_mode == LLAMA_SPLIT_MODE_TENSOR) {150        throw common_params_fit_exception("llama_params_fit is not implemented for SPLIT_MODE_TENSOR, abort");151    }152    constexpr int64_t MiB = 1024*1024;153    typedef std::vector<llama_device_memory_data> dmds_t;154    const llama_model_params default_mparams = llama_model_default_params();155 156    std::vector<ggml_backend_dev_t> devs;157    uint32_t hp_ngl = 0; // hparams.n_gpu_layers158    uint32_t hp_nct = 0; // hparams.n_ctx_train159    uint32_t hp_nex = 0; // hparams.n_expert160 161    // step 1: get data for default parameters and check whether any changes are necessary in the first place162 163    LOG_INF("%s: getting device memory data for initial parameters:\n", __func__);164    const dmds_t dmds_full = common_get_device_memory_data(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);165    const size_t nd = devs.size(); // number of devices166 167    std::vector<int64_t> margins; // this function uses int64_t rather than size_t for memory sizes to more conveniently handle deficits168    margins.reserve(nd);169    if (nd == 0) {170        margins.push_back(margins_s[0]);171    } else {172        for (size_t id = 0; id < nd; id++) {173            margins.push_back(margins_s[id]);174        }175    }176 177    std::vector<std::string> dev_names;178    {179        dev_names.reserve(nd);180        size_t max_length = 0;181        for (const auto & dev : devs) {182            std::string name = ggml_backend_dev_name(dev);183            name += " (";184            name += ggml_backend_dev_description(dev);185            name += ")";186            dev_names.push_back(name);187            max_length = std::max(max_length, name.length());188        }189        for (std::string & dn : dev_names) {190            dn.insert(dn.end(), max_length - dn.length(), ' ');191        }192    }193 194    int64_t sum_free            = 0;195    int64_t sum_projected_free  = 0;196    int64_t sum_projected_used  = 0;197    int64_t sum_projected_model = 0;198    std::vector<int64_t> projected_free_per_device;199    projected_free_per_device.reserve(nd);200 201    if (nd == 0) {202        sum_projected_used = dmds_full.back().mb.total();203        sum_free           = dmds_full.back().total;204        sum_projected_free = sum_free - sum_projected_used;205        LOG_INF("%s: projected to use %" PRId64 " MiB of host memory vs. %" PRId64 " MiB of total host memory\n",206            __func__, sum_projected_used/MiB, sum_free/MiB);207        if (sum_projected_free >= margins[0]) {208            LOG_INF("%s: will leave %" PRId64 " >= %" PRId64 " MiB of system memory, no changes needed\n",209                __func__, sum_projected_free/MiB, margins[0]/MiB);210            return;211        }212    } else {213        if (nd > 1) {214            LOG_INF("%s: projected memory use with initial parameters [MiB]:\n", __func__);215        }216        for (size_t id = 0; id < nd; id++) {217            const llama_device_memory_data & dmd = dmds_full[id];218 219            const int64_t projected_used = dmd.mb.total();220            const int64_t projected_free = dmd.free - projected_used;221            projected_free_per_device.push_back(projected_free);222 223            sum_free            += dmd.free;224            sum_projected_used  += projected_used;225            sum_projected_free  += projected_free;226            sum_projected_model += dmd.mb.model;227 228            if (nd > 1) {229                LOG_INF("%s:   - %s: %6" PRId64 " total, %6" PRId64 " used, %6" PRId64 " free vs. target of %6" PRId64 "\n",230                    __func__, dev_names[id].c_str(), dmd.total/MiB, projected_used/MiB, projected_free/MiB, margins[id]/MiB);231            }232        }233        assert(sum_free >= 0 && sum_projected_used >= 0);234        LOG_INF("%s: projected to use %" PRId64 " MiB of device memory vs. %" PRId64 " MiB of free device memory\n",235            __func__, sum_projected_used/MiB, sum_free/MiB);236        if (nd == 1) {237            if (projected_free_per_device[0] >= margins[0]) {238                LOG_INF("%s: will leave %" PRId64 " >= %" PRId64 " MiB of free device memory, no changes needed\n",239                    __func__, projected_free_per_device[0]/MiB, margins[0]/MiB);240                return;241            }242        } else {243            bool changes_needed = false;244            for (size_t id = 0; id < nd; id++) {245                if (projected_free_per_device[id] < margins[id]) {246                    changes_needed = true;247                    break;248                }249            }250            if (!changes_needed) {251                LOG_INF("%s: targets for free memory can be met on all devices, no changes needed\n", __func__);252                return;253            }254        }255    }256 257    // step 2: try reducing memory use by reducing the context size258 259    {260        int64_t global_surplus = sum_projected_free;261        if (nd == 0) {262            global_surplus -= margins[0];263        } else {264            for (size_t id = 0; id < nd; id++) {265                global_surplus -= margins[id];266            }267        }268        if (global_surplus < 0) {269            if (nd <= 1) {270                LOG_INF("%s: cannot meet free memory target of %" PRId64 " MiB, need to reduce device memory by %" PRId64 " MiB\n",271                    __func__, margins[0]/MiB, -global_surplus/MiB);272            } else {273                LOG_INF(274                    "%s: cannot meet free memory targets on all devices, need to use %" PRId64 " MiB less in total\n",275                    __func__, -global_surplus/MiB);276            }277            if (cparams->n_ctx == 0) {278                if (hp_nct > n_ctx_min) {279                    int64_t sum_used_target = sum_free;280                    if (nd == 0) {281                        sum_used_target -= margins[0];282                    } else {283                        for (size_t id = 0; id < nd; id++) {284                            sum_used_target -= margins[id];285                        }286                    }287                    if (nd > 1) {288                        // for multiple devices we need to be more conservative in terms of how much context we think can fit:289                        //   - for dense models only whole layers can be assigned to devices290                        //   - for MoE models only whole tensors can be assigned to devices, which we estimate to be <= 1/3 of a layer291                        //   - on average we expect a waste of 0.5 layers/tensors per device292                        //   - use slightly more than the expected average for nd devices to be safe293                        const int64_t model_per_layer = sum_projected_model / std::min(uint32_t(mparams->n_gpu_layers), hp_ngl);294                        sum_used_target -= (nd + 1) * model_per_layer / (hp_nex == 0 ? 2 : 6);295                    }296 297                    int64_t sum_projected_used_min_ctx = 0;298                    cparams->n_ctx = n_ctx_min;299                    const dmds_t dmds_min_ctx = common_get_device_memory_data(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);300                    if (nd == 0) {301                        sum_projected_used_min_ctx = dmds_min_ctx.back().mb.total();302                    } else {303                        for (size_t id = 0; id < nd; id++) {304                            sum_projected_used_min_ctx += dmds_min_ctx[id].mb.total();305                        }306                    }307                    if (sum_used_target > sum_projected_used_min_ctx) {308                        // linear interpolation between minimum and maximum context size:309                        cparams->n_ctx += (hp_nct - n_ctx_min) * (sum_used_target - sum_projected_used_min_ctx)310                            / (sum_projected_used - sum_projected_used_min_ctx);311                        cparams->n_ctx = std::max(cparams->n_ctx - cparams->n_ctx % 256, n_ctx_min); // round down context for CUDA backend312 313                        const int64_t bytes_per_ctx = (sum_projected_used - sum_projected_used_min_ctx) / (hp_nct - n_ctx_min);314                        const int64_t memory_reduction = (hp_nct - cparams->n_ctx) * bytes_per_ctx;315                        LOG_INF("%s: context size reduced from %" PRIu32 " to %" PRIu32 " -> need %" PRId64 " MiB less memory in total\n",316                            __func__, hp_nct, cparams->n_ctx, memory_reduction/MiB);317                        if (nd <= 1) {318                            LOG_INF("%s: entire model can be fit by reducing context\n", __func__);319                            return;320                        }321                        LOG_INF("%s: entire model should be fit across devices by reducing context\n", __func__);322                    } else {323                        const int64_t memory_reduction = sum_projected_used - sum_projected_used_min_ctx;324                        LOG_INF("%s: context size reduced from %" PRIu32 " to %" PRIu32 " -> need %" PRId64 " MiB less memory in total\n",325                            __func__, hp_nct, cparams->n_ctx, memory_reduction/MiB);326                    }327                } else {328                    if (n_ctx_min == UINT32_MAX) {329                        LOG_INF("%s: user has requested full context size of %" PRIu32 " -> no change\n", __func__, hp_nct);330                    } else {331                        LOG_INF("%s: default model context size is %" PRIu32 " which is <= the min. context size of %" PRIu32 " -> no change\n",332                            __func__, hp_nct, n_ctx_min);333                    }334                }335            } else {336                LOG_INF("%s: context size set by user to %" PRIu32 " -> no change\n", __func__, cparams->n_ctx);337            }338        }339    }340    if (nd == 0) {341        throw common_params_fit_exception("was unable to fit model into system memory by reducing context, abort");342    }343 344    if (mparams->n_gpu_layers != default_mparams.n_gpu_layers) {345        throw common_params_fit_exception("n_gpu_layers already set by user to " + std::to_string(mparams->n_gpu_layers) + ", abort");346    }347    if (nd > 1) {348        if (!tensor_split) {349            throw common_params_fit_exception("did not provide a buffer to write the tensor_split to, abort");350        }351        if (mparams->tensor_split) {352            for (size_t id = 0; id < nd; id++) {353                if (mparams->tensor_split[id] != 0.0f) {354                    throw common_params_fit_exception("model_params::tensor_split already set by user, abort");355                }356            }357        }358        if (mparams->split_mode == LLAMA_SPLIT_MODE_ROW) {359            throw common_params_fit_exception("changing weight allocation for LLAMA_SPLIT_MODE_ROW not implemented, abort");360        }361    }362    if (!tensor_buft_overrides) {363        throw common_params_fit_exception("did not provide buffer to set tensor_buft_overrides, abort");364    }365    if (mparams->tensor_buft_overrides && (mparams->tensor_buft_overrides->pattern || mparams->tensor_buft_overrides->buft)) {366        throw common_params_fit_exception("model_params::tensor_buft_overrides already set by user, abort");367    }368 369    // step 3: iteratively fill the back to front with "dense" layers370    //   - for a dense model simply fill full layers, giving each device a contiguous slice of the model371    //   - for a MoE model, same as dense model but with all MoE tensors in system memory372 373    // utility function that returns a static C string matching the tensors for a specific layer index and layer fraction:374    auto get_overflow_pattern = [&](const size_t il, const common_layer_fraction_t lf) -> const char * {375        constexpr size_t n_strings = 1000;376        if (il >= n_strings) {377            throw std::runtime_error("at most " + std::to_string(n_strings) + " model layers are supported");378        }379        switch (lf) {380            case LAYER_FRACTION_ATTN: {381                static std::array<std::string, n_strings> patterns;382                if (patterns[il].empty()) {383                    patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(gate|up|gate_up|down).*";384                }385                return patterns[il].c_str();386            }387            case LAYER_FRACTION_UP: {388                static std::array<std::string, n_strings> patterns;389                if (patterns[il].empty()) {390                    patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(gate|gate_up|down).*";391                }392                return patterns[il].c_str();393            }394            case LAYER_FRACTION_GATE: {395                static std::array<std::string, n_strings> patterns;396                if (patterns[il].empty()) {397                    patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_down.*";398                }399                return patterns[il].c_str();400            }401            case LAYER_FRACTION_MOE: {402                static std::array<std::string, n_strings> patterns;403                if (patterns[il].empty()) {404                    patterns[il] = "blk\\." + std::to_string(il) + "\\.ffn_(up|down|gate_up|gate)_(ch|)exps";405                }406                return patterns[il].c_str();407            }408            default:409                GGML_ABORT("fatal error");410        }411    };412 413    struct ngl_t {414        uint32_t n_layer = 0; // number of total layers415        uint32_t n_part  = 0; // number of partial layers, <= n_layer416 417        // for the first partial layer varying parts can overflow, all further layers use LAYER_FRACTION_MOE:418        common_layer_fraction_t overflow_type = LAYER_FRACTION_MOE;419 420        uint32_t n_full() const {421            assert(n_layer >= n_part);422            return n_layer - n_part;423        }424    };425 426    const size_t ntbo = llama_max_tensor_buft_overrides();427 428    // utility function to set n_gpu_layers and tensor_split429    auto set_ngl_tensor_split_tbo = [&](430            const std::vector<ngl_t> & ngl_per_device,431            const std::vector<ggml_backend_buffer_type_t> & overflow_bufts,432            llama_model_params & mparams) {433        mparams.n_gpu_layers = 0;434        for (size_t id = 0; id < nd; id++) {435            mparams.n_gpu_layers += ngl_per_device[id].n_layer;436            if (nd > 1) {437                tensor_split[id] = ngl_per_device[id].n_layer;438            }439        }440        assert(uint32_t(mparams.n_gpu_layers) <= hp_ngl + 1);441        uint32_t il0 = hp_ngl + 1 - mparams.n_gpu_layers; // start index for tensor buft overrides442 443        mparams.tensor_split = tensor_split;444 445        size_t itbo = 0;446        for (size_t id = 0; id < nd; id++) {447            il0 += ngl_per_device[id].n_full();448            for (uint32_t il = il0; il < il0 + ngl_per_device[id].n_part; il++) {449                if (itbo + 1 >= ntbo) {450                    tensor_buft_overrides[itbo].pattern = nullptr;451                    tensor_buft_overrides[itbo].buft    = nullptr;452                    itbo++;453                    mparams.tensor_buft_overrides = tensor_buft_overrides;454                    throw common_params_fit_exception("llama_max_tensor_buft_overrides() == "455                        + std::to_string(ntbo) + " is insufficient for model");456                }457                tensor_buft_overrides[itbo].pattern = get_overflow_pattern(il, il == il0 ? ngl_per_device[id].overflow_type : LAYER_FRACTION_MOE);458                tensor_buft_overrides[itbo].buft = il == il0 ? overflow_bufts[id] : ggml_backend_cpu_buffer_type();459                itbo++;460            }461            il0 += ngl_per_device[id].n_part;462        }463        tensor_buft_overrides[itbo].pattern = nullptr;464        tensor_buft_overrides[itbo].buft    = nullptr;465        itbo++;466        mparams.tensor_buft_overrides = tensor_buft_overrides;467    };468 469    // utility function that returns the memory use per device for given numbers of layers per device470    auto get_memory_for_layers = [&](471            const char * func_name,472            const std::vector<ngl_t> & ngl_per_device,473            const std::vector<ggml_backend_buffer_type_t> & overflow_bufts) -> std::vector<int64_t> {474        llama_model_params mparams_copy = *mparams;475        set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, mparams_copy);476 477        const dmds_t dmd_nl = common_get_device_memory_data(478            path_model, &mparams_copy, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);479 480        LOG_INF("%s: memory for test allocation by device:\n", func_name);481        for (size_t id = 0; id < nd; id++) {482            const ngl_t & n = ngl_per_device[id];483            LOG_INF(484                "%s: id=%zu, n_layer=%2" PRIu32 ", n_part=%2" PRIu32 ", overflow_type=%d, mem=%6" PRId64 " MiB\n",485                func_name, id, n.n_layer, n.n_part, int(n.overflow_type), dmd_nl[id].mb.total()/MiB);486        }487 488        std::vector<int64_t> ret;489        ret.reserve(nd);490        for (size_t id = 0; id < nd; id++) {491            ret.push_back(dmd_nl[id].mb.total());492        }493        return ret;494    };495 496    int64_t global_surplus_cpu_moe = 0;497    if (hp_nex > 0) {498        const static std::string pattern_moe_all = "blk\\.\\d+\\.ffn_(up|down|gate_up|gate)_(ch|)exps"; // matches all MoE tensors499        ggml_backend_buffer_type_t cpu_buft = ggml_backend_cpu_buffer_type();500        tensor_buft_overrides[0] = {pattern_moe_all.c_str(), cpu_buft};501        tensor_buft_overrides[1] = {nullptr, nullptr};502        mparams->tensor_buft_overrides = tensor_buft_overrides;503 504        LOG_INF("%s: getting device memory data with all MoE tensors moved to system memory:\n", __func__);505        const dmds_t dmds_cpu_moe = common_get_device_memory_data(506            path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, log_level);507 508        for (size_t id = 0; id < nd; id++) {509            global_surplus_cpu_moe += dmds_cpu_moe[id].free;510            global_surplus_cpu_moe -= int64_t(dmds_cpu_moe[id].mb.total()) + margins[id];511        }512 513        if (global_surplus_cpu_moe > 0) {514            LOG_INF("%s: with only dense weights in device memory there is a total surplus of %" PRId64 " MiB\n",515                __func__, global_surplus_cpu_moe/MiB);516        } else {517            LOG_INF("%s: with only dense weights in device memory there is still a total deficit of %" PRId64 " MiB\n",518                __func__, -global_surplus_cpu_moe/MiB);519        }520 521        // reset522        tensor_buft_overrides[0] = {nullptr, nullptr};523        mparams->tensor_buft_overrides = tensor_buft_overrides;524    }525 526    std::vector<int64_t> targets; // maximum acceptable memory use per device527    targets.reserve(nd);528    for (size_t id = 0; id < nd; id++) {529        targets.push_back(dmds_full[id].free - margins[id]);530        LOG_INF("%s: id=%zu, target=%" PRId64 " MiB\n", __func__, id, targets[id]/MiB);531    }532 533    std::vector<ggml_backend_buffer_type_t> overflow_bufts; // which bufts the first partial layer of a device overflows to:534    overflow_bufts.reserve(nd);535    for (size_t id = 0; id < nd; id++) {536        overflow_bufts.push_back(ggml_backend_cpu_buffer_type());537    }538 539    std::vector<ngl_t> ngl_per_device(nd);540    std::vector<int64_t> mem = get_memory_for_layers(__func__, ngl_per_device, overflow_bufts);541 542    // optimize the number of layers per device using the method of false position:543    //   - ngl_per_device has 0 layers for each device, lower bound544    //   - try a "high" configuration where a device is given all unassigned layers545    //   - interpolate the memory use / layer between low and high linearly to get a guess where it meets our target546    //   - check memory use of our guess, replace either the low or high bound547    //   - once we only have a difference of a single layer, stop and return the lower bound that just barely still fits548    //   - the last device has the output layer, which cannot be a partial layer549    if (hp_nex == 0) {550        LOG_INF("%s: filling dense layers back-to-front:\n", __func__);551    } else {552        LOG_INF("%s: filling dense-only layers back-to-front:\n", __func__);553    }554    for (int id = nd - 1; id >= 0; id--) {555        uint32_t n_unassigned = hp_ngl + 1;556        for (size_t jd = id + 1; jd < nd; ++jd) {557            assert(n_unassigned >= ngl_per_device[jd].n_layer);558            n_unassigned -= ngl_per_device[jd].n_layer;559        }560 561        std::vector<ngl_t> ngl_per_device_high = ngl_per_device;562        ngl_per_device_high[id].n_layer = n_unassigned;563        if (hp_nex > 0) {564            ngl_per_device_high[id].n_part = size_t(id) < nd - 1 ? ngl_per_device_high[id].n_layer : ngl_per_device_high[id].n_layer - 1;565        }566        if (ngl_per_device_high[id].n_layer > 0) {567            std::vector<int64_t> mem_high = get_memory_for_layers(__func__, ngl_per_device_high, overflow_bufts);568            if (mem_high[id] > targets[id]) {569                assert(ngl_per_device_high[id].n_layer > ngl_per_device[id].n_layer);570                uint32_t delta = ngl_per_device_high[id].n_layer - ngl_per_device[id].n_layer;571                LOG_INF("%s: start filling device %" PRIu32 ", delta=%" PRIu32 "\n", __func__, id, delta);572                while (delta > 1) {573                    uint32_t step_size = int64_t(delta) * (targets[id] - mem[id]) / (mem_high[id] - mem[id]);574                    step_size = std::max(step_size, uint32_t(1));575                    step_size = std::min(step_size, delta - 1);576 577                    std::vector<ngl_t> ngl_per_device_test = ngl_per_device;578                    ngl_per_device_test[id].n_layer += step_size;579                    if (hp_nex) {580                        ngl_per_device_test[id].n_part += size_t(id) == nd - 1 && ngl_per_device_test[id].n_part == 0 ?581                            step_size - 1 : step_size; // the first layer is the output layer which must always be full582                    }583                    const std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts);584 585                    if (mem_test[id] <= targets[id]) {586                        ngl_per_device = ngl_per_device_test;587                        mem            = mem_test;588                        LOG_INF("%s: set ngl_per_device[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device[id].n_layer);589                    } else {590                        ngl_per_device_high = ngl_per_device_test;591                        mem_high            = mem_test;592                        LOG_INF("%s: set ngl_per_device_high[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device_high[id].n_layer);593                    }594                    delta = ngl_per_device_high[id].n_layer - ngl_per_device[id].n_layer;595                }596            } else {597                assert(ngl_per_device_high[id].n_layer == n_unassigned);598                ngl_per_device = ngl_per_device_high;599                mem            = mem_high;600                LOG_INF("%s: set ngl_per_device[%d].n_layer=%" PRIu32 "\n", __func__, id, ngl_per_device[id].n_layer);601            }602        }603 604        const int64_t projected_margin = dmds_full[id].free - mem[id];605        LOG_INF(606            "%s:   - %s: %2" PRIu32 " layers, %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",607            __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, mem[id]/MiB, projected_margin/MiB);608    }609    if (hp_nex == 0 || global_surplus_cpu_moe <= 0) {610        set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, *mparams);611        return;612    }613 614    // step 4: for a MoE model where all dense tensors fit,615    //     convert the dense-only layers in the back to full layers in the front until all devices are full616    // essentially the same procedure as for the dense-only layers except front-to-back617    // also, try fitting at least part of one more layer to reduce waste for "small" GPUs with e.g. 24 GiB VRAM618 619    size_t id_dense_start = nd;620    for (int id = nd - 1; id >= 0; id--) {621        if (ngl_per_device[id].n_layer > 0) {622            id_dense_start = id;623            continue;624        }625        break;626    }627    assert(id_dense_start < nd);628 629    LOG_INF("%s: converting dense-only layers to full layers and filling them front-to-back with overflow to next device/system memory:\n", __func__);630    for (size_t id = 0; id <= id_dense_start && id_dense_start < nd; id++) {631        std::vector<ngl_t> ngl_per_device_high = ngl_per_device;632        for (size_t jd = id_dense_start; jd < nd; jd++) {633            const uint32_t n_layer_move = jd < nd - 1 ? ngl_per_device_high[jd].n_layer : ngl_per_device_high[jd].n_layer - 1;634            ngl_per_device_high[id].n_layer += n_layer_move;635            ngl_per_device_high[jd].n_layer -= n_layer_move;636            ngl_per_device_high[jd].n_part = 0;637        }638        size_t id_dense_start_high = nd - 1;639        std::vector<int64_t> mem_high = get_memory_for_layers(__func__, ngl_per_device_high, overflow_bufts);640 641        if (mem_high[id] > targets[id]) {642            assert(ngl_per_device_high[id].n_full() >= ngl_per_device[id].n_full());643            uint32_t delta = ngl_per_device_high[id].n_full() - ngl_per_device[id].n_full();644            while (delta > 1) {645                uint32_t step_size = int64_t(delta) * (targets[id] - mem[id]) / (mem_high[id] - mem[id]);646                step_size = std::max(step_size, uint32_t(1));647                step_size = std::min(step_size, delta - 1);648 649                std::vector<ngl_t> ngl_per_device_test = ngl_per_device;650                size_t id_dense_start_test = id_dense_start;651                uint32_t n_converted_test = 0;652                for (;id_dense_start_test < nd; id_dense_start_test++) {653                    const uint32_t n_convert_jd = std::min(step_size - n_converted_test, ngl_per_device_test[id_dense_start_test].n_part);654                    ngl_per_device_test[id_dense_start_test].n_layer -= n_convert_jd;655                    ngl_per_device_test[id_dense_start_test].n_part -= n_convert_jd;656                    ngl_per_device_test[id].n_layer += n_convert_jd;657                    n_converted_test += n_convert_jd;658 659                    if (ngl_per_device_test[id_dense_start_test].n_part > 0) {660                        break;661                    }662                }663                const std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts);664 665                if (mem_test[id] <= targets[id]) {666                    ngl_per_device = ngl_per_device_test;667                    mem            = mem_test;668                    id_dense_start = id_dense_start_test;669                    LOG_INF("%s: set ngl_per_device[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start=%zu\n",670                        __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);671                } else {672                    ngl_per_device_high = ngl_per_device_test;673                    mem_high            = mem_test;674                    id_dense_start_high = id_dense_start_test;675                    LOG_INF("%s: set ngl_per_device_high[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start_high=%zu\n",676                        __func__, id, ngl_per_device_high[id].n_layer, ngl_per_device_high[id].n_part, id_dense_start_high);677                }678                assert(ngl_per_device_high[id].n_full() >= ngl_per_device[id].n_full());679                delta = ngl_per_device_high[id].n_full() - ngl_per_device[id].n_full();680            }681        } else {682            ngl_per_device = ngl_per_device_high;683            mem            = mem_high;684            id_dense_start = id_dense_start_high;685            LOG_INF("%s: set ngl_per_device[%zu].(n_layer, n_part)=(%" PRIu32 ", %" PRIu32 "), id_dense_start=%zu\n",686                __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);687        }688 689        // try to fit at least part of one more layer690        if (ngl_per_device[id_dense_start].n_layer > (id < nd - 1 ? 0 : 1)) {691            std::vector<ngl_t> ngl_per_device_test = ngl_per_device;692            size_t id_dense_start_test = id_dense_start;693            ngl_per_device_test[id_dense_start_test].n_layer--;694            ngl_per_device_test[id_dense_start_test].n_part--;695            ngl_per_device_test[id].n_layer++;696            ngl_per_device_test[id].n_part++;697            if (ngl_per_device_test[id_dense_start_test].n_part == 0) {698                id_dense_start_test++;699            }700            ngl_per_device_test[id].overflow_type = LAYER_FRACTION_UP;701            std::vector<ggml_backend_buffer_type_t> overflow_bufts_test = overflow_bufts;702            if (id < nd - 1) {703                overflow_bufts_test[id] = ggml_backend_dev_buffer_type(devs[id + 1]);704            }705            LOG_INF("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_UP\n", __func__);706            std::vector<int64_t> mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);707            if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {708                ngl_per_device = ngl_per_device_test;709                overflow_bufts = overflow_bufts_test;710                mem            = mem_test;711                id_dense_start = id_dense_start_test;712                LOG_INF("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", UP), id_dense_start=%zu\n",713                    __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);714 715                ngl_per_device_test[id].overflow_type = LAYER_FRACTION_GATE;716                LOG_INF("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_GATE\n", __func__);717                mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);718                if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {719                    ngl_per_device = ngl_per_device_test;720                    overflow_bufts = overflow_bufts_test;721                    mem            = mem_test;722                    id_dense_start = id_dense_start_test;723                    LOG_INF("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", GATE), id_dense_start=%zu\n",724                        __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);725                }726            } else {727                ngl_per_device_test[id].overflow_type = LAYER_FRACTION_ATTN;728                LOG_INF("%s: trying to fit one extra layer with overflow_type=LAYER_FRACTION_ATTN\n", __func__);729                mem_test = get_memory_for_layers(__func__, ngl_per_device_test, overflow_bufts_test);730                if (mem_test[id] < targets[id] && (id + 1 == nd || mem_test[id + 1] < targets[id + 1])) {731                    ngl_per_device = ngl_per_device_test;732                    overflow_bufts = overflow_bufts_test;733                    mem            = mem_test;734                    id_dense_start = id_dense_start_test;735                    LOG_INF("%s: set ngl_per_device[%zu].(n_layer, n_part, overflow_type)=(%" PRIu32 ", %" PRIu32 ", ATTN), id_dense_start=%zu\n",736                        __func__, id, ngl_per_device[id].n_layer, ngl_per_device[id].n_part, id_dense_start);737                }738            }739        }740 741        const int64_t projected_margin = dmds_full[id].free - mem[id];742        LOG_INF(743            "%s:   - %s: %2" PRIu32 " layers (%2" PRIu32 " overflowing), %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",744            __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, ngl_per_device[id].n_part, mem[id]/MiB, projected_margin/MiB);745    }746 747    // print info for devices that were not changed during the conversion from dense only to full layers:748    for (size_t id = id_dense_start + 1; id < nd; id++) {749        const int64_t projected_margin = dmds_full[id].free - mem[id];750        LOG_INF(751            "%s:   - %s: %2" PRIu32 " layers (%2" PRIu32 " overflowing), %6" PRId64 " MiB used, %6" PRId64 " MiB free\n",752            __func__, dev_names[id].c_str(), ngl_per_device[id].n_layer, ngl_per_device[id].n_part, mem[id]/MiB, projected_margin/MiB);753    }754 755    set_ngl_tensor_split_tbo(ngl_per_device, overflow_bufts, *mparams);756}757 758enum common_params_fit_status common_fit_params(759        const char * path_model,760        llama_model_params * mparams,761        llama_context_params * cparams,762        float * tensor_split,763        llama_model_tensor_buft_override * tensor_buft_overrides,764        size_t * margins,765        uint32_t n_ctx_min,766        ggml_log_level log_level) {767    const int64_t t0_us = llama_time_us();768    common_params_fit_status status = COMMON_PARAMS_FIT_STATUS_SUCCESS;769    try {770        common_params_fit_impl(path_model, mparams, cparams, tensor_split, tensor_buft_overrides, margins, n_ctx_min, log_level);771        LOG_INF("%s: successfully fit params to free device memory\n", __func__);772    } catch (const common_params_fit_exception & e) {773        LOG_WRN("%s: failed to fit params to free device memory: %s\n", __func__, e.what());774        status = COMMON_PARAMS_FIT_STATUS_FAILURE;775    } catch (const std::runtime_error & e) {776        LOG_ERR("%s: encountered an error while trying to fit params to free device memory: %s\n", __func__, e.what());777        status = COMMON_PARAMS_FIT_STATUS_ERROR;778    }779    const int64_t t1_us = llama_time_us();780    LOG_INF("%s: fitting params to free memory took %.2f seconds\n", __func__, (t1_us - t0_us) * 1e-6);781    return status;782}783 784void common_memory_breakdown_print(const struct llama_context * ctx) {785    //const auto & devices = ctx->get_model().devices;786    const auto * model = llama_get_model(ctx);787 788    std::vector<ggml_backend_dev_t> devices;789    for (int i = 0; i < llama_model_n_devices(model); i++) {790        devices.push_back(llama_model_get_device(model, i));791    }792 793    llama_memory_breakdown memory_breakdown = llama_get_memory_breakdown(ctx);794 795    std::vector<std::array<std::string, 9>> table_data;796    table_data.reserve(devices.size());797    const std::string template_header = "%s: | %s | %s   %s    %s   %s   %s   %s    %s |\n";798    const std::string template_gpu    = "%s: | %s | %s = %s + (%s = %s + %s + %s) + %s |\n";799    const std::string template_other  = "%s: | %s | %s   %s    %s = %s + %s + %s    %s |\n";800 801    table_data.push_back({template_header, "memory breakdown [MiB]", "total", "free", "self", "model", "context", "compute", "unaccounted"});802 803    constexpr size_t MiB = 1024 * 1024;804    const std::vector<std::string> desc_prefixes_strip = {"NVIDIA ", "GeForce ", "Tesla ", "AMD ", "Radeon ", "Instinct "};805 806    // track seen buffer types to avoid double counting:807    std::set<ggml_backend_buffer_type_t> seen_buffer_types;808 809    // accumulative memory breakdown for each device and for host:810    std::vector<llama_memory_breakdown_data> mb_dev(devices.size());811    llama_memory_breakdown_data              mb_host;812 813    for (const auto & buft_mb : memory_breakdown) {814        ggml_backend_buffer_type_t          buft = buft_mb.first;815        const llama_memory_breakdown_data & mb   = buft_mb.second;816        if (ggml_backend_buft_is_host(buft)) {817            mb_host.model   += mb.model;818            mb_host.context += mb.context;819            mb_host.compute += mb.compute;820            seen_buffer_types.insert(buft);821            continue;822        }823        ggml_backend_dev_t dev = ggml_backend_buft_get_device(buft);824        if (dev) {825            int i_dev = -1;826            for (size_t i = 0; i < devices.size(); i++) {827                if (devices[i] == dev) {828                    i_dev = i;829                    break;830                }831            }832            if (i_dev != -1) {833                mb_dev[i_dev].model   += mb.model;834                mb_dev[i_dev].context += mb.context;835                mb_dev[i_dev].compute += mb.compute;836                seen_buffer_types.insert(buft);837                continue;838            }839        }840    }841 842    // print memory breakdown for each device:843    for (size_t i = 0; i < devices.size(); i++) {844        ggml_backend_dev_t dev = devices[i];845        llama_memory_breakdown_data mb = mb_dev[i];846 847        const std::string name = ggml_backend_dev_name(dev);848        std::string desc = ggml_backend_dev_description(dev);849        for (const std::string & prefix : desc_prefixes_strip) {850            if (desc.length() >= prefix.length() && desc.substr(0, prefix.length()) == prefix) {851                desc = desc.substr(prefix.length());852            }853        }854 855        size_t free, total;856        ggml_backend_dev_memory(dev, &free, &total);857 858        const size_t self = mb.model + mb.context + mb.compute;859        const size_t unaccounted = total - self - free;860 861        table_data.push_back({862            template_gpu,863            "  - " + name + " (" + desc + ")",864            std::to_string(total / MiB),865            std::to_string(free / MiB),866            std::to_string(self / MiB),867            std::to_string(mb.model / MiB),868            std::to_string(mb.context / MiB),869            std::to_string(mb.compute / MiB),870            std::to_string(unaccounted / MiB)});871    }872 873    // print memory breakdown for host:874    {875        const size_t self = mb_host.model + mb_host.context + mb_host.compute;876        table_data.push_back({877            template_other,878            "  - Host",879            "", // total880            "", // free881            std::to_string(self / MiB),882            std::to_string(mb_host.model / MiB),883            std::to_string(mb_host.context / MiB),884            std::to_string(mb_host.compute / MiB),885            ""}); // unaccounted886    }887 888    // print memory breakdown for all remaining buffer types:889    for (const auto & buft_mb : memory_breakdown) {890        ggml_backend_buffer_type_t          buft = buft_mb.first;891        const llama_memory_breakdown_data & mb   = buft_mb.second;892        if (seen_buffer_types.count(buft) == 1) {893            continue;894        }895        const std::string name = ggml_backend_buft_name(buft);896        const size_t self = mb.model + mb.context + mb.compute;897        table_data.push_back({898            template_other,899            "  - " + name,900            "", // total901            "", // free902            std::to_string(self / MiB),903            std::to_string(mb.model / MiB),904            std::to_string(mb.context / MiB),905            std::to_string(mb.compute / MiB),906            ""}); // unaccounted907        seen_buffer_types.insert(buft);908    }909 910    for (size_t j = 1; j < table_data[0].size(); j++) {911        size_t max_len = 0;912        for (const auto & td : table_data) {913            max_len = std::max(max_len, td[j].length());914        }915        for (auto & td : table_data) {916            td[j].insert(j == 1 ? td[j].length() : 0, max_len - td[j].length(), ' ');917        }918    }919    for (const auto & td : table_data) {920        LOG_INF(td[0].c_str(),921            __func__, td[1].c_str(), td[2].c_str(), td[3].c_str(), td[4].c_str(), td[5].c_str(),922            td[6].c_str(), td[7].c_str(), td[8].c_str());923    }924}925 926void common_fit_print(927        const char * path_model,928        llama_model_params * mparams,929        llama_context_params * cparams) {930    std::vector<ggml_backend_dev_t> devs;931    uint32_t hp_ngl = 0; // hparams.n_gpu_layers932    uint32_t hp_nct = 0; // hparams.n_ctx_train933    uint32_t hp_nex = 0; // hparams.n_expert934 935    auto dmd = common_get_device_memory_data(path_model, mparams, cparams, devs, hp_ngl, hp_nct, hp_nex, GGML_LOG_LEVEL_ERROR);936    GGML_ASSERT(dmd.size() == devs.size() + 1);937 938    for (size_t id = 0; id < devs.size(); id++) {939        printf("%s ",  ggml_backend_dev_name(devs[id]));940        printf("%zu ", dmd[id].mb.model/1024/1024);941        printf("%zu ", dmd[id].mb.context/1024/1024);942        printf("%zu ", dmd[id].mb.compute/1024/1024);943        printf("\n");944    }945 946    printf("Host ");947    printf("%zu ", dmd.back().mb.model/1024/1024);948    printf("%zu ", dmd.back().mb.context/1024/1024);949    printf("%zu ", dmd.back().mb.compute/1024/1024);950    printf("\n");951}952