Team Ai
Datasetpublic

echodict/llama.cpp

version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes604downloads
llama-quant.cpp1403 linesDownload Raw Back to src
1#include "llama-impl.h"2#include "llama-model.h"3#include "llama-model-loader.h"4#include "llama-ext.h"5 6#include <algorithm>7#include <cmath>8#include <cstring>9#include <cinttypes>10#include <fstream>11#include <mutex>12#include <regex>13#include <thread>14#include <unordered_map>15 16// result of parsing --tensor-type option17// (changes to this struct must be reflected in tools/quantize/quantize.cpp)18struct tensor_type_option {19    std::string name;20    ggml_type type = GGML_TYPE_COUNT;21};22 23// tensor categorization - used to avoid repeated string matching in quantization logic.24// this is different from LLM_TN - we want broad categories, not specific tensor names per arch.25enum class tensor_category {26    TOKEN_EMBD,27    ATTENTION_Q,28    ATTENTION_V,29    ATTENTION_K,30    ATTENTION_QKV,31    ATTENTION_KV_B,32    ATTENTION_OUTPUT,33    FFN_UP,34    FFN_GATE,35    FFN_DOWN,36    OUTPUT,37    OTHER38};39 40static void zeros(std::ofstream & file, size_t n) {41    char zero = 0;42    for (size_t i = 0; i < n; ++i) {43        file.write(&zero, 1);44    }45}46 47static std::string remap_layer(const std::string & orig_name, const std::vector<int> & prune, std::map<int, std::string> & mapped, int & next_id) {48    if (prune.empty()) {49        return orig_name;50    }51 52    static const std::regex pattern(R"(blk\.(\d+)\.)");53    if (std::smatch match; std::regex_search(orig_name, match, pattern)) {54        const int blk = std::stoi(match[1]);55        std::string new_name = orig_name;56 57        if (mapped.count(blk)) {58            // Already mapped, do nothing59        } else if (std::find(prune.begin(), prune.end(), blk) != prune.end()) {60            mapped[blk] = "";61        } else if (blk < prune.front()) {62            mapped[blk] = std::to_string(blk);63            next_id = blk + 1;64        } else {65            mapped[blk] = std::to_string(next_id);66            ++next_id;67        }68 69        return mapped[blk].empty() ? mapped[blk] : new_name.replace(match.position(1), match.length(1), mapped[blk]);70    }71 72    return orig_name;73}74 75static std::string remap_imatrix(const std::string & orig_name, const std::map<int, std::string> & mapped) {76    if (mapped.empty()) {77        return orig_name;78    }79 80    static const std::regex pattern(R"(blk\.(\d+)\.)");81    if (std::smatch match; std::regex_search(orig_name, match, pattern)) {82        const std::string blk(match[1]);83        std::string new_name = orig_name;84 85        for (const auto & p : mapped) {86            if (p.second == blk) {87                return new_name.replace(match.position(1), match.length(1), std::to_string(p.first));88            }89        }90        GGML_ABORT("\n%s: imatrix mapping error for %s\n", __func__, orig_name.c_str());91    }92 93    return orig_name;94}95 96//97// helper functions for tensor name matching98//99 100static bool tensor_name_match_token_embd(const char * tensor_name) {101    return std::strcmp(tensor_name, "token_embd.weight") == 0 ||102           std::strcmp(tensor_name, "per_layer_token_embd.weight") == 0;103}104 105static bool tensor_name_match_output_weight(const char * tensor_name) {106    return std::strcmp(tensor_name, "output.weight") == 0;107}108 109//110// tensor categorization for quantization111//112// (this is different from LLM_TN - we want broad categories, not specific tensor names per arch)113//114 115static tensor_category tensor_get_category(const std::string & tensor_name) {116    if (tensor_name_match_output_weight(tensor_name.c_str())) {117        return tensor_category::OUTPUT;118    }119    if (tensor_name_match_token_embd(tensor_name.c_str())) {120        return tensor_category::TOKEN_EMBD;121    }122    if (tensor_name.find("attn_qkv.weight") != std::string::npos) {123        return tensor_category::ATTENTION_QKV;124    }125    if (tensor_name.find("attn_kv_b.weight") != std::string::npos) {126        return tensor_category::ATTENTION_KV_B;127    }128    if (tensor_name.find("attn_v.weight") != std::string::npos) {129        return tensor_category::ATTENTION_V;130    }131    if (tensor_name.find("attn_k.weight") != std::string::npos) {132        return tensor_category::ATTENTION_K;133    }134    if (tensor_name.find("attn_q.weight") != std::string::npos) {135        return tensor_category::ATTENTION_Q;136    }137    if (tensor_name.find("attn_output.weight") != std::string::npos) {138        return tensor_category::ATTENTION_OUTPUT;139    }140    if (tensor_name.find("ffn_up") != std::string::npos) {141        return tensor_category::FFN_UP;142    }143    if (tensor_name.find("ffn_gate") != std::string::npos) {144        return tensor_category::FFN_GATE;145    }146    if (tensor_name.find("ffn_down") != std::string::npos) {147        return tensor_category::FFN_DOWN;148    }149    return tensor_category::OTHER;150}151 152// check if category is for attention-v-like tensors (more sensitive to quantization)153static bool category_is_attn_v(tensor_category cat) {154    return cat == tensor_category::ATTENTION_V     ||155           cat == tensor_category::ATTENTION_QKV   ||156           cat == tensor_category::ATTENTION_KV_B;157}158 159//160// quantization state161//162 163struct quantize_state_impl {164    const llama_model                 & model;165    const llama_model_quantize_params * params;166 167    int n_attention_wv = 0;168    int n_ffn_down     = 0;169    int n_ffn_gate     = 0;170    int n_ffn_up       = 0;171    int i_attention_wv = 0;172    int i_ffn_down     = 0;173    int i_ffn_gate     = 0;174    int i_ffn_up       = 0;175 176    int n_fallback    = 0;177 178    bool has_imatrix = false;179 180    // used to figure out if a model has tied embeddings (tok_embd shares weights with output)181    bool has_tied_embeddings = true; // assume tied until we see output.weight182 183    // tensor type override patterns (compiled once, used twice)184    std::vector<std::pair<std::regex, ggml_type>> tensor_type_patterns;185 186    quantize_state_impl(const llama_model & model, const llama_model_quantize_params * params):187        model(model), params(params)188    {189        // compile regex patterns once - they are expensive190        if (params->tt_overrides) {191            for (const auto * p = params->tt_overrides; p->pattern != nullptr; p++) {192                tensor_type_patterns.emplace_back(std::regex(p->pattern), p->type);193            }194        }195    }196};197 198// per-tensor metadata, computed in the preliminary loop and used in the main loop199struct tensor_metadata {200    std::string     name;201    ggml_type       target_type;202    tensor_category category;203    std::string     remapped_imatrix_name;204    bool            allows_quantization;205    bool            requires_imatrix;206};207 208//209// dequantization210//211 212static void llama_tensor_dequantize_impl(213    ggml_tensor * tensor, std::vector<no_init<float>> & output, std::vector<std::thread> & workers,214    const size_t nelements, const int nthread215) {216    if (output.size() < nelements) {217        output.resize(nelements);218    }219    float * f32_output = (float *) output.data();220 221    const ggml_type_traits * qtype = ggml_get_type_traits(tensor->type);222    if (ggml_is_quantized(tensor->type)) {223        if (qtype->to_float == NULL) {224            throw std::runtime_error(format("type %s unsupported for integer quantization: no dequantization available", ggml_type_name(tensor->type)));225        }226    } else if (tensor->type != GGML_TYPE_F16 &&227               tensor->type != GGML_TYPE_BF16) {228        throw std::runtime_error(format("cannot dequantize/convert tensor type %s", ggml_type_name(tensor->type)));229    }230 231    if (nthread < 2) {232        if (tensor->type == GGML_TYPE_F16) {233            ggml_fp16_to_fp32_row((ggml_fp16_t *)tensor->data, f32_output, nelements);234        } else if (tensor->type == GGML_TYPE_BF16) {235            ggml_bf16_to_fp32_row((ggml_bf16_t *)tensor->data, f32_output, nelements);236        } else if (ggml_is_quantized(tensor->type)) {237            qtype->to_float(tensor->data, f32_output, nelements);238        } else {239            GGML_ABORT("fatal error"); // unreachable240        }241        return;242    }243 244    size_t block_size;245    if (tensor->type == GGML_TYPE_F16 ||246        tensor->type == GGML_TYPE_BF16) {247        block_size = 1;248    } else {249        block_size = (size_t)ggml_blck_size(tensor->type);250    }251 252    size_t block_size_bytes = ggml_type_size(tensor->type);253 254    GGML_ASSERT(nelements % block_size == 0);255    size_t nblocks = nelements / block_size;256    size_t blocks_per_thread = nblocks / nthread;257    size_t spare_blocks = nblocks - (blocks_per_thread * nthread); // if blocks aren't divisible by thread count258 259    size_t in_buff_offs = 0;260    size_t out_buff_offs = 0;261 262    for (int tnum = 0; tnum < nthread; tnum++) {263        size_t thr_blocks = blocks_per_thread + (tnum == nthread - 1 ? spare_blocks : 0); // num blocks for this thread264        size_t thr_elems = thr_blocks * block_size; // number of elements for this thread265        size_t thr_block_bytes = thr_blocks * block_size_bytes; // number of input bytes for this thread266 267        auto compute = [qtype] (ggml_type typ, uint8_t * inbuf, float * outbuf, int nels) {268            if (typ == GGML_TYPE_F16) {269                ggml_fp16_to_fp32_row((ggml_fp16_t *)inbuf, outbuf, nels);270            } else if (typ == GGML_TYPE_BF16) {271                ggml_bf16_to_fp32_row((ggml_bf16_t *)inbuf, outbuf, nels);272            } else {273                qtype->to_float(inbuf, outbuf, nels);274            }275        };276        workers.emplace_back(compute, tensor->type, (uint8_t *) tensor->data + in_buff_offs, f32_output + out_buff_offs, thr_elems);277        in_buff_offs += thr_block_bytes;278        out_buff_offs += thr_elems;279    }280    for (auto & w : workers) { w.join(); }281    workers.clear();282}283 284//285// do we allow this tensor to be quantized?286//287 288static bool tensor_allows_quantization(const llama_model_quantize_params * params, llm_arch arch, const ggml_tensor * tensor) {289    // trivial checks first -- no string ops needed290    if (params->only_copy)       return false;291 292    // quantize only 2D and 3D tensors (experts)293    if (ggml_n_dims(tensor) < 2) return false;294 295    const std::string name = ggml_get_name(tensor);296 297    // This used to be a regex, but <regex> has an extreme cost to compile times.298    bool quantize = name.rfind("weight") == name.size() - 6; // ends with 'weight'?299 300    // do not quantize norm tensors301    quantize &= name.find("_norm.weight") == std::string::npos;302 303    quantize &= params->quantize_output_tensor || name != "output.weight";304 305    // do not quantize expert gating tensors306    // NOTE: can't use LLM_TN here because the layer number is not known307    quantize &= name.find("ffn_gate_inp.weight") == std::string::npos;308 309    // these are very small (e.g. 4x4)310    quantize &= name.find("altup")  == std::string::npos;311    quantize &= name.find("laurel") == std::string::npos;312 313    // these are not too big so keep them as it is314    quantize &= name.find("per_layer_model_proj") == std::string::npos;315 316    // do not quantize positional embeddings and token types (BERT)317    quantize &= name != LLM_TN(arch)(LLM_TENSOR_POS_EMBD,    "weight");318    quantize &= name != LLM_TN(arch)(LLM_TENSOR_TOKEN_TYPES, "weight");319 320    // do not quantize Mamba/Kimi's small conv1d weights321    // NOTE: can't use LLM_TN here because the layer number is not known322    quantize &= name.find("ssm_conv1d") == std::string::npos;323    quantize &= name.find("shortconv.conv.weight") == std::string::npos;324 325    // do not quantize RWKV's small yet 2D weights326    quantize &= name.find("time_mix_first.weight") == std::string::npos;327    quantize &= name.find("time_mix_w0.weight") == std::string::npos;328    quantize &= name.find("time_mix_w1.weight") == std::string::npos;329    quantize &= name.find("time_mix_w2.weight") == std::string::npos;330    quantize &= name.find("time_mix_v0.weight") == std::string::npos;331    quantize &= name.find("time_mix_v1.weight") == std::string::npos;332    quantize &= name.find("time_mix_v2.weight") == std::string::npos;333    quantize &= name.find("time_mix_a0.weight") == std::string::npos;334    quantize &= name.find("time_mix_a1.weight") == std::string::npos;335    quantize &= name.find("time_mix_a2.weight") == std::string::npos;336    quantize &= name.find("time_mix_g1.weight") == std::string::npos;337    quantize &= name.find("time_mix_g2.weight") == std::string::npos;338    quantize &= name.find("time_mix_decay_w1.weight") == std::string::npos;339    quantize &= name.find("time_mix_decay_w2.weight") == std::string::npos;340    quantize &= name.find("time_mix_lerp_fused.weight") == std::string::npos;341 342    // do not quantize relative position bias (T5)343    quantize &= name.find("attn_rel_b.weight") == std::string::npos;344 345    // do not quantize specific multimodal tensors346    quantize &= name.find(".position_embd") == std::string::npos;347    quantize &= name.find("sam.pos_embd")   == std::string::npos;348    quantize &= name.find("sam.neck.")      == std::string::npos;349    quantize &= name.find("sam.net_")       == std::string::npos;350    quantize &= name.find(".rel_pos")       == std::string::npos;351    quantize &= name.find(".patch_embd")    == std::string::npos;352    quantize &= name.find(".patch_merger")  == std::string::npos;353 354    return quantize;355}356 357//358// tensor type selection359//360 361// incompatible tensor shapes are handled here - fallback to a compatible type362static ggml_type tensor_type_fallback(quantize_state_impl & qs, const ggml_tensor * t, const ggml_type target_type) {363    ggml_type return_type = target_type;364 365    const int64_t ncols = t->ne[0];366    const int64_t qk_k = ggml_blck_size(target_type);367 368    if (ncols % qk_k != 0) { // this tensor's shape is incompatible with this quant369        LLAMA_LOG_WARN("warning: %-36s - ncols %6" PRId64 " not divisible by %3" PRId64 " (required for type %7s) ",370                        t->name, ncols, qk_k, ggml_type_name(target_type));371        ++qs.n_fallback;372 373        switch (target_type) {374            // types on the left: block size 256375            case GGML_TYPE_IQ1_S:376            case GGML_TYPE_IQ1_M:377            case GGML_TYPE_IQ2_XXS:378            case GGML_TYPE_IQ2_XS:379            case GGML_TYPE_IQ2_S:380            case GGML_TYPE_IQ3_XXS:381            case GGML_TYPE_IQ3_S:   // types on the right: block size 32382            case GGML_TYPE_IQ4_XS:  return_type = GGML_TYPE_IQ4_NL; break;383            case GGML_TYPE_Q2_K:384            case GGML_TYPE_Q3_K:385            case GGML_TYPE_TQ1_0:386            case GGML_TYPE_TQ2_0:   return_type = GGML_TYPE_Q4_0;   break;387            case GGML_TYPE_Q4_K:    return_type = GGML_TYPE_Q5_0;   break;388            case GGML_TYPE_Q5_K:    return_type = GGML_TYPE_Q5_1;   break;389            case GGML_TYPE_Q6_K:    return_type = GGML_TYPE_Q8_0;   break;390            default:391                throw std::runtime_error(format("no tensor type fallback is defined for type %s",392                                                ggml_type_name(target_type)));393        }394        if (ncols % ggml_blck_size(return_type) != 0) {395            //396            // the fallback return type is still not compatible for this tensor!397            //398            // most likely, this tensor's first dimension is not divisible by 32.399            // this is very rare. we can either abort the quantization, or400            // fallback to F16 / F32.401            //402            LLAMA_LOG_WARN("(WARNING: must use F16 due to unusual shape) ");403            return_type = GGML_TYPE_F16;404        }405        LLAMA_LOG_WARN("-> falling back to %7s\n", ggml_type_name(return_type));406    }407    return return_type;408}409 410// internal standard logic for selecting the target tensor type based on tensor category, ftype, and model arch411static ggml_type llama_tensor_get_type_impl(quantize_state_impl & qs, ggml_type new_type, const ggml_tensor * tensor, llama_ftype ftype, tensor_category category) {412    const std::string name = ggml_get_name(tensor);413 414    // TODO: avoid hardcoded tensor names - use the TN_* constants415    const llm_arch arch = qs.model.arch;416 417    auto use_more_bits = [](int i_layer, int n_layers) -> bool {418        return i_layer < n_layers/8 || i_layer >= 7*n_layers/8 || (i_layer - n_layers/8)%3 == 2;419    };420    const int n_expert = std::max(1, (int)qs.model.hparams.n_expert);421    auto layer_info = [n_expert] (int i_layer, int n_layer, const char * name) {422        if (n_expert > 1) {423            // Believe it or not, "experts" in the FFN of Mixtral-8x7B are not consecutive, but occasionally randomly424            // sprinkled in the model. Hence, simply dividing i_ffn_down by n_expert does not work425            // for getting the current layer as I initially thought, and we need to resort to parsing the426            // tensor name.427            if (sscanf(name, "blk.%d.", &i_layer) != 1) {428                throw std::runtime_error(format("Failed to determine layer for tensor %s", name));429            }430            if (i_layer < 0 || i_layer >= n_layer) {431                throw std::runtime_error(format("Bad layer %d for tensor %s. Must be in [0, %d)", i_layer, name, n_layer));432            }433        }434        return std::make_pair(i_layer, n_layer);435    };436 437    // for arches that share the same tensor between the token embeddings and the output, we quantize the token embeddings438    // with the quantization of the output tensor439    if (category == tensor_category::OUTPUT || (qs.has_tied_embeddings && category == tensor_category::TOKEN_EMBD)) {440        if (qs.params->output_tensor_type < GGML_TYPE_COUNT) {441            new_type = qs.params->output_tensor_type;442        } else {443            const int64_t nx = tensor->ne[0];444            const int64_t qk_k = ggml_blck_size(new_type);445 446            if (ftype == LLAMA_FTYPE_MOSTLY_MXFP4_MOE) {447                new_type = GGML_TYPE_Q8_0;448            }449            else if (arch == LLM_ARCH_FALCON || nx % qk_k != 0) {450                new_type = GGML_TYPE_Q8_0;451            }452            else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS ||453                     ftype == LLAMA_FTYPE_MOSTLY_IQ1_S   || ftype == LLAMA_FTYPE_MOSTLY_IQ2_S  || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M   ||454                     ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {455                new_type = GGML_TYPE_Q5_K;456            }457            else if (new_type != GGML_TYPE_Q8_0) {458                new_type = GGML_TYPE_Q6_K;459            }460        }461    } else if (ftype == LLAMA_FTYPE_MOSTLY_MXFP4_MOE) {462        // MoE   tensors -> MXFP4463        // other tensors -> Q8_0464        if (tensor->ne[2] > 1) {465            new_type = GGML_TYPE_MXFP4;466        } else {467            new_type = GGML_TYPE_Q8_0;468        }469    } else if (category == tensor_category::TOKEN_EMBD) {470        if (qs.params->token_embedding_type < GGML_TYPE_COUNT) {471            new_type = qs.params->token_embedding_type;472        } else {473            if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS ||474                ftype == LLAMA_FTYPE_MOSTLY_IQ1_S   || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {475                new_type = GGML_TYPE_Q2_K;476            }477            else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M) {478                new_type = GGML_TYPE_IQ3_S;479            }480            else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {481                new_type = GGML_TYPE_IQ3_S;482            }483            else if (ftype == LLAMA_FTYPE_MOSTLY_TQ1_0 || ftype == LLAMA_FTYPE_MOSTLY_TQ2_0) {484                new_type = GGML_TYPE_Q4_K;485            }486        }487    } else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ1_S ||488               ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M    || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {489        if (category_is_attn_v(category)) {490            if (qs.model.hparams.n_gqa() >= 4 || qs.model.hparams.n_expert >= 4) new_type = GGML_TYPE_Q4_K;491            else new_type = ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ? GGML_TYPE_IQ3_S : GGML_TYPE_Q2_K;492            ++qs.i_attention_wv;493        }494        else if (qs.model.hparams.n_expert == 8 && category == tensor_category::ATTENTION_K) {495            new_type = GGML_TYPE_Q4_K;496        }497        else if (category == tensor_category::FFN_DOWN) {498            if (qs.i_ffn_down < qs.n_ffn_down/8) {499                new_type = ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ? GGML_TYPE_IQ3_S : GGML_TYPE_Q2_K;500            }501            ++qs.i_ffn_down;502        }503        else if (category == tensor_category::ATTENTION_OUTPUT) {504            if (qs.model.hparams.n_expert == 8) {505                new_type = GGML_TYPE_Q5_K;506            } else {507                if (ftype == LLAMA_FTYPE_MOSTLY_IQ1_S || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) new_type = GGML_TYPE_IQ2_XXS;508                else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M) new_type = GGML_TYPE_IQ3_S;509            }510        }511    } else if (category_is_attn_v(category)) {512        if      (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) {513            new_type = qs.model.hparams.n_gqa() >= 4 ? GGML_TYPE_Q4_K : GGML_TYPE_Q3_K;514        }515        else if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S && qs.model.hparams.n_gqa() >= 4) {516            new_type = GGML_TYPE_Q4_K;517        }518        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {519            new_type = qs.model.hparams.n_gqa() >= 4 ? GGML_TYPE_Q4_K : !qs.has_imatrix ? GGML_TYPE_IQ3_S : GGML_TYPE_IQ3_XXS;520        }521        else if ((ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_S) && qs.model.hparams.n_gqa() >= 4) {522            new_type = GGML_TYPE_Q4_K;523        }524        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M) {525            new_type = GGML_TYPE_Q4_K;526        }527        else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {528            new_type = qs.i_attention_wv < 2 ? GGML_TYPE_Q5_K : GGML_TYPE_Q4_K;529        }530        else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) new_type = GGML_TYPE_Q5_K;531        else if ((ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) && qs.model.hparams.n_gqa() >= 4) {532            new_type = GGML_TYPE_Q5_K;533        }534        else if ((ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) &&535                use_more_bits(qs.i_attention_wv, qs.n_attention_wv)) new_type = GGML_TYPE_Q6_K;536        else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S && qs.i_attention_wv < 4) new_type = GGML_TYPE_Q5_K;537        if (qs.model.type == LLM_TYPE_70B) {538            // In the 70B model we have 8 heads sharing the same attn_v weights. As a result, the attn_v.weight tensor is539            // 8x smaller compared to attn_q.weight. Hence, we can get a nice boost in quantization accuracy with540            // nearly negligible increase in model size by quantizing this tensor with more bits:541            if (new_type == GGML_TYPE_Q3_K || new_type == GGML_TYPE_Q4_K) new_type = GGML_TYPE_Q5_K;542        }543        if (qs.model.hparams.n_expert == 8) {544            // for the 8-expert model, bumping this to Q8_0 trades just ~128MB545            // TODO: explore better strategies546            new_type = GGML_TYPE_Q8_0;547        }548        ++qs.i_attention_wv;549    } else if (category == tensor_category::ATTENTION_K) {550        if (qs.model.hparams.n_expert == 8) {551            // for the 8-expert model, bumping this to Q8_0 trades just ~128MB552            // TODO: explore better strategies553            new_type = GGML_TYPE_Q8_0;554        }555        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS) {556            new_type = GGML_TYPE_IQ3_XXS;557        }558        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {559            new_type = GGML_TYPE_IQ2_S;560        }561    } else if (category == tensor_category::ATTENTION_Q) {562        if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS) {563            new_type = GGML_TYPE_IQ3_XXS;564        }565        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {566            new_type = GGML_TYPE_IQ2_S;567        }568    } else if (category == tensor_category::FFN_DOWN) {569        auto info = layer_info(qs.i_ffn_down, qs.n_ffn_down, name.c_str());570        int i_layer = info.first, n_layer = info.second;571        if      (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) new_type = GGML_TYPE_Q3_K;572        else if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S) {573            if (i_layer < n_layer/8) new_type = GGML_TYPE_Q4_K;574        }575        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS && !qs.has_imatrix) {576            new_type = i_layer < n_layer/8 ? GGML_TYPE_Q4_K : GGML_TYPE_Q3_K;577        }578        else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {579            new_type = i_layer < n_layer/16 ? GGML_TYPE_Q5_K580                     : arch != LLM_ARCH_FALCON || use_more_bits(i_layer, n_layer) ? GGML_TYPE_Q4_K581                     : GGML_TYPE_Q3_K;582        }583        else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M && (i_layer < n_layer/8 ||584                    (qs.model.hparams.n_expert == 8 && use_more_bits(i_layer, n_layer)))) {585            new_type = GGML_TYPE_Q4_K;586        }587        else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {588            new_type = arch == LLM_ARCH_FALCON ? GGML_TYPE_Q4_K : GGML_TYPE_Q5_K;589        }590        else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) {591            if (arch == LLM_ARCH_FALCON) {592                new_type = i_layer < n_layer/16 ? GGML_TYPE_Q6_K :593                           use_more_bits(i_layer, n_layer) ? GGML_TYPE_Q5_K : GGML_TYPE_Q4_K;594            } else {595                if (use_more_bits(i_layer, n_layer)) new_type = GGML_TYPE_Q6_K;596            }597        }598        else if (i_layer < n_layer/8 && (ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) && !qs.has_imatrix) {599            new_type = GGML_TYPE_Q5_K;600        }601        else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M && use_more_bits(i_layer, n_layer)) new_type = GGML_TYPE_Q6_K;602        else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S && arch != LLM_ARCH_FALCON && i_layer < n_layer/8) {603            new_type = GGML_TYPE_Q5_K;604        }605        else if ((ftype == LLAMA_FTYPE_MOSTLY_Q4_0 || ftype == LLAMA_FTYPE_MOSTLY_Q5_0)606                && qs.has_imatrix && i_layer < n_layer/8) {607            // Guard against craziness in the first few ffn_down layers that can happen even with imatrix for Q4_0/Q5_0.608            // We only do it when an imatrix is provided because a) we want to make sure that one can always get the609            // same quantization as before imatrix stuff, and b) Q4_1/Q5_1 do go crazy on ffn_down without an imatrix.610            new_type = ftype == LLAMA_FTYPE_MOSTLY_Q4_0 ? GGML_TYPE_Q4_1 : GGML_TYPE_Q5_1;611        }612        ++qs.i_ffn_down;613    } else if (category == tensor_category::ATTENTION_OUTPUT) {614        if (arch != LLM_ARCH_FALCON) {615            if (qs.model.hparams.n_expert == 8) {616                if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K   || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS ||617                    ftype == LLAMA_FTYPE_MOSTLY_Q3_K_S || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M  || ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL  ||618                    ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S || ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M  || ftype == LLAMA_FTYPE_MOSTLY_IQ3_S  ||619                    ftype == LLAMA_FTYPE_MOSTLY_IQ3_M  || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) {620                    new_type = GGML_TYPE_Q5_K;621                }622            } else {623                if      (ftype == LLAMA_FTYPE_MOSTLY_Q2_K   ) new_type = GGML_TYPE_Q3_K;624                else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) new_type = GGML_TYPE_IQ3_S;625                else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M ) new_type = GGML_TYPE_Q4_K;626                else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L ) new_type = GGML_TYPE_Q5_K;627                else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M  ) new_type = GGML_TYPE_Q4_K;628            }629        } else {630            if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) new_type = GGML_TYPE_Q4_K;631        }632    }633    else if (category == tensor_category::ATTENTION_QKV) {634        if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L || ftype == LLAMA_FTYPE_MOSTLY_IQ3_M) {635            new_type = GGML_TYPE_Q4_K;636        }637        else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) new_type = GGML_TYPE_Q5_K;638        else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) new_type = GGML_TYPE_Q6_K;639    }640    else if (category == tensor_category::FFN_GATE) {641        auto info = layer_info(qs.i_ffn_gate, qs.n_ffn_gate, name.c_str());642        int i_layer = info.first, n_layer = info.second;643        if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS && (i_layer >= n_layer/8 && i_layer < 7*n_layer/8)) {644            new_type = GGML_TYPE_IQ3_XXS;645        }646        ++qs.i_ffn_gate;647    }648    else if (category == tensor_category::FFN_UP) {649        auto info = layer_info(qs.i_ffn_up, qs.n_ffn_up, name.c_str());650        int i_layer = info.first, n_layer = info.second;651        if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS && (i_layer >= n_layer/8 && i_layer < 7*n_layer/8)) {652            new_type = GGML_TYPE_IQ3_XXS;653        }654        ++qs.i_ffn_up;655    }656 657    return new_type;658}659 660// outer wrapper: determine the ggml_type that this tensor should be quantized to661static ggml_type llama_tensor_get_type(quantize_state_impl & qs, const llama_model_quantize_params * params, const ggml_tensor * tensor, ggml_type default_type, const tensor_metadata & tm) {662    if (!tensor_allows_quantization(params, qs.model.arch, tensor)) {663        return tensor->type;664    }665    if (params->token_embedding_type < GGML_TYPE_COUNT && tm.category == tensor_category::TOKEN_EMBD) {666        return params->token_embedding_type;667    }668    if (params->output_tensor_type < GGML_TYPE_COUNT && tm.category == tensor_category::OUTPUT) {669        return params->output_tensor_type;670    }671 672    ggml_type new_type = default_type;673 674    // get more optimal quantization type based on the tensor shape, layer, etc.675    if (!params->pure && ggml_is_quantized(default_type)) {676        // if the user provided tensor types - use those677        bool manual = false;678        if (!qs.tensor_type_patterns.empty()) {679            const std::string tensor_name(tensor->name);680            for (const auto & [pattern, qtype] : qs.tensor_type_patterns) {681                if (std::regex_search(tensor_name, pattern)) {682                    if (qtype != new_type) {683                        LLAMA_LOG_WARN("%s: %-36s - applying manual override: %s -> %s\n",684                                       __func__, tensor_name.c_str(), ggml_type_name(new_type), ggml_type_name(qtype));685                        new_type = qtype;686                        manual = true;687                        break;688                    }689                }690            }691        }692 693        // if not manual - use the standard logic for choosing the quantization type based on the selected mixture694        if (!manual) {695            new_type = llama_tensor_get_type_impl(qs, new_type, tensor, params->ftype, tm.category);696        }697 698        // incompatible tensor shapes are handled here - fallback to a compatible type699        new_type = tensor_type_fallback(qs, tensor, new_type);700    }701 702    return new_type;703}704 705//706// quantization implementation707//708 709static size_t llama_tensor_quantize_impl(enum ggml_type new_type, const float * f32_data, void * new_data, const int64_t chunk_size, int64_t nrows, int64_t n_per_row, const float * imatrix, std::vector<std::thread> & workers, const int nthread) {710    if (nthread < 2) {711        // single-thread712        size_t new_size = ggml_quantize_chunk(new_type, f32_data, new_data, 0, nrows, n_per_row, imatrix);713        if (!ggml_validate_row_data(new_type, new_data, new_size)) {714            throw std::runtime_error("quantized data validation failed");715        }716        return new_size;717    }718 719    std::mutex mutex;720    int64_t counter = 0;721    size_t new_size = 0;722    bool valid = true;723    auto compute = [&mutex, &counter, &new_size, &valid, new_type, f32_data, new_data, chunk_size,724            nrows, n_per_row, imatrix]() {725        const int64_t nrows_per_chunk = chunk_size / n_per_row;726        size_t local_size = 0;727        while (true) {728            std::unique_lock<std::mutex> lock(mutex);729            int64_t first_row = counter; counter += nrows_per_chunk;730            if (first_row >= nrows) {731                if (local_size > 0) {732                    new_size += local_size;733                }734                break;735            }736            lock.unlock();737            const int64_t this_nrow = std::min(nrows - first_row, nrows_per_chunk);738            size_t this_size = ggml_quantize_chunk(new_type, f32_data, new_data, first_row * n_per_row, this_nrow, n_per_row, imatrix);739            local_size += this_size;740 741            // validate the quantized data742            const size_t row_size  = ggml_row_size(new_type, n_per_row);743            void * this_data = (char *) new_data + first_row * row_size;744            if (!ggml_validate_row_data(new_type, this_data, this_size)) {745                std::unique_lock<std::mutex> lock(mutex);746                valid = false;747                break;748            }749        }750    };751    for (int it = 0; it < nthread - 1; ++it) {752        workers.emplace_back(compute);753    }754    compute();755    for (auto & w : workers) { w.join(); }756    workers.clear();757    if (!valid) {758        throw std::runtime_error("quantized data validation failed");759    }760    return new_size;761}762 763//764// imatrix requirement check765//766 767static bool tensor_requires_imatrix(const char * tensor_name, const ggml_type dst_type, const llama_ftype ftype) {768    if (tensor_name_match_token_embd(tensor_name) || tensor_name_match_output_weight(tensor_name)) {769        return false;770    }771    switch (dst_type) {772        case GGML_TYPE_IQ3_XXS:773        case GGML_TYPE_IQ2_XXS:774        case GGML_TYPE_IQ2_XS:775        case GGML_TYPE_IQ2_S:776        case GGML_TYPE_IQ1_M:777        case GGML_TYPE_IQ1_S:778            return true;779        case GGML_TYPE_Q2_K:780            // as a general rule, the k-type quantizations don't require imatrix data.781            // the only exception is Q2_K tensors that are part of a Q2_K_S file.782            return ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S;783        default:784            return false;785    }786}787 788//789// given a file type, get the default tensor type790//791 792ggml_type llama_ftype_get_default_type(llama_ftype ftype) {793    switch (ftype) {794        case LLAMA_FTYPE_MOSTLY_Q4_0: return GGML_TYPE_Q4_0;795        case LLAMA_FTYPE_MOSTLY_Q4_1: return GGML_TYPE_Q4_1;796        case LLAMA_FTYPE_MOSTLY_Q5_0: return GGML_TYPE_Q5_0;797        case LLAMA_FTYPE_MOSTLY_Q5_1: return GGML_TYPE_Q5_1;798        case LLAMA_FTYPE_MOSTLY_Q8_0: return GGML_TYPE_Q8_0;799        case LLAMA_FTYPE_MOSTLY_F16:  return GGML_TYPE_F16;800        case LLAMA_FTYPE_MOSTLY_BF16: return GGML_TYPE_BF16;801        case LLAMA_FTYPE_ALL_F32:     return GGML_TYPE_F32;802        case LLAMA_FTYPE_MOSTLY_Q1_0: return GGML_TYPE_Q1_0;803 804        case LLAMA_FTYPE_MOSTLY_MXFP4_MOE: return GGML_TYPE_MXFP4;805 806        // K-quants807        case LLAMA_FTYPE_MOSTLY_Q2_K_S:808        case LLAMA_FTYPE_MOSTLY_Q2_K:    return GGML_TYPE_Q2_K;809        case LLAMA_FTYPE_MOSTLY_IQ3_XS:  return GGML_TYPE_IQ3_S;810        case LLAMA_FTYPE_MOSTLY_Q3_K_S:811        case LLAMA_FTYPE_MOSTLY_Q3_K_M:812        case LLAMA_FTYPE_MOSTLY_Q3_K_L:  return GGML_TYPE_Q3_K;813        case LLAMA_FTYPE_MOSTLY_Q4_K_S:814        case LLAMA_FTYPE_MOSTLY_Q4_K_M:  return GGML_TYPE_Q4_K;815        case LLAMA_FTYPE_MOSTLY_Q5_K_S:816        case LLAMA_FTYPE_MOSTLY_Q5_K_M:  return GGML_TYPE_Q5_K;817        case LLAMA_FTYPE_MOSTLY_Q6_K:    return GGML_TYPE_Q6_K;818        case LLAMA_FTYPE_MOSTLY_TQ1_0:   return GGML_TYPE_TQ1_0;819        case LLAMA_FTYPE_MOSTLY_TQ2_0:   return GGML_TYPE_TQ2_0;820        case LLAMA_FTYPE_MOSTLY_IQ2_XXS: return GGML_TYPE_IQ2_XXS;821        case LLAMA_FTYPE_MOSTLY_IQ2_XS:  return GGML_TYPE_IQ2_XS;822        case LLAMA_FTYPE_MOSTLY_IQ2_S:   return GGML_TYPE_IQ2_XS;823        case LLAMA_FTYPE_MOSTLY_IQ2_M:   return GGML_TYPE_IQ2_S;824        case LLAMA_FTYPE_MOSTLY_IQ3_XXS: return GGML_TYPE_IQ3_XXS;825        case LLAMA_FTYPE_MOSTLY_IQ1_S:   return GGML_TYPE_IQ1_S;826        case LLAMA_FTYPE_MOSTLY_IQ1_M:   return GGML_TYPE_IQ1_M;827        case LLAMA_FTYPE_MOSTLY_IQ4_NL:  return GGML_TYPE_IQ4_NL;828        case LLAMA_FTYPE_MOSTLY_IQ4_XS:  return GGML_TYPE_IQ4_XS;829        case LLAMA_FTYPE_MOSTLY_IQ3_S:830        case LLAMA_FTYPE_MOSTLY_IQ3_M:   return GGML_TYPE_IQ3_S;831 832        default: return GGML_TYPE_COUNT;833    }834}835 836 837static void init_quantize_state_counters(quantize_state_impl & qs, std::vector<tensor_metadata> & metadata) {838    for (auto & tm : metadata) {839        tensor_category cat = tensor_get_category(tm.name);840        tm.category = cat;841 842        if (category_is_attn_v(cat)) {843            ++qs.n_attention_wv;844        }845 846        if (cat == tensor_category::OUTPUT) {847            qs.has_tied_embeddings = false;848        }849    }850    qs.n_ffn_down = qs.n_ffn_gate = qs.n_ffn_up = (int)qs.model.hparams.n_layer;851}852 853//854// main quantization driver855//856 857static void llama_model_quantize_impl(const std::string & fname_inp, const std::string & fname_out, const llama_model_quantize_params * params) {858    llama_ftype ftype = params->ftype;859 860    int nthread = params->nthread;861 862    if (nthread <= 0) {863        nthread = std::thread::hardware_concurrency();864    }865 866    ggml_type default_type = llama_ftype_get_default_type(ftype);867    if (default_type == GGML_TYPE_COUNT) {868        throw std::runtime_error(format("invalid output file type %d\n", ftype));869    }870 871    // mmap consistently increases speed on Linux, and also increases speed on Windows with872    // hot cache. It may cause a slowdown on macOS, possibly related to free memory.873#if defined(__linux__) || defined(_WIN32)874    constexpr bool use_mmap = true;875#else876    constexpr bool use_mmap = false;877#endif878 879    const llama_model_kv_override * kv_overrides = params->kv_overrides;880    std::vector<std::string> splits = {};881    llama_model_loader ml(/*metadata*/ nullptr, /*set_tensor_data*/ nullptr, /*set_tensor_data_ud*/ nullptr,882        fname_inp, splits, /*file*/ nullptr, use_mmap, /*use_direct_io*/ false, /*check_tensors*/ true, /*no_alloc*/ false, kv_overrides, nullptr);883    ml.init_mappings(false); // no prefetching884 885    llama_model model(llama_model_default_params());886 887    model.load_arch   (ml);888    model.load_hparams(ml);889    model.load_stats  (ml);890 891    quantize_state_impl qs(model, params);892 893    if (params->only_copy) {894        ftype = ml.ftype;895    }896    std::unordered_map<std::string, std::vector<float>> i_data;897    const std::unordered_map<std::string, std::vector<float>> * imatrix_data = nullptr;898    if (params->imatrix) {899        for (const llama_model_imatrix_data * p = params->imatrix; p->name != nullptr; p++) {900            i_data.emplace(p->name, std::vector<float>(p->data, p->data + p->size));901        }902        imatrix_data = & i_data;903        if (imatrix_data) {904            LLAMA_LOG_INFO("\n%s: have importance matrix data with %d entries\n",905                           __func__, (int)imatrix_data->size());906            qs.has_imatrix = true;907            // check imatrix for nans or infs908            for (const auto & kv : *imatrix_data) {909                for (float f : kv.second) {910                    if (!std::isfinite(f)) {911                        throw std::runtime_error(format("imatrix contains non-finite value %f\n", f));912                    }913                }914            }915        }916    }917 918    const size_t align = GGUF_DEFAULT_ALIGNMENT;919    gguf_context_ptr ctx_out { gguf_init_empty() };920 921    std::vector<int> prune_list = {};922    if (params->prune_layers) {923        for (const int32_t * p = params->prune_layers; * p != -1; p++) {924            prune_list.push_back(* p);925        }926    }927 928    // copy the KV pairs from the input file929    gguf_set_kv     (ctx_out.get(), ml.metadata);930    gguf_set_val_u32(ctx_out.get(), "general.quantization_version", GGML_QNT_VERSION); // TODO: use LLM_KV931    gguf_set_val_u32(ctx_out.get(), "general.file_type", ftype); // TODO: use LLM_KV932 933    // Remove split metadata934    gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_NO).c_str());935    gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_COUNT).c_str());936    gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_TENSORS_COUNT).c_str());937 938    if (params->kv_overrides) {939        for (const llama_model_kv_override * o = params->kv_overrides; o->key[0] != 0; ++o) {940            if (o->tag == LLAMA_KV_OVERRIDE_TYPE_FLOAT) {941                gguf_set_val_f32(ctx_out.get(), o->key, o->val_f64);942            } else if (o->tag == LLAMA_KV_OVERRIDE_TYPE_INT) {943                // Setting type to UINT32. See https://github.com/ggml-org/llama.cpp/pull/14182 for context944                gguf_set_val_u32(ctx_out.get(), o->key, (uint32_t)std::abs(o->val_i64));945            } else if (o->tag == LLAMA_KV_OVERRIDE_TYPE_BOOL) {946                gguf_set_val_bool(ctx_out.get(), o->key, o->val_bool);947            } else if (o->tag == LLAMA_KV_OVERRIDE_TYPE_STR) {948                gguf_set_val_str(ctx_out.get(), o->key, o->val_str);949            } else {950                LLAMA_LOG_WARN("%s: unknown KV override type for key %s\n", __func__, o->key);951            }952        }953    }954 955    std::map<int, std::string> mapped;956    int blk_id = 0;957 958    // make a list of weights959    std::vector<const llama_model_loader::llama_tensor_weight *> tensors;960    tensors.reserve(ml.weights_map.size());961    for (const auto & it : ml.weights_map) {962        const std::string remapped_name(remap_layer(it.first, prune_list, mapped, blk_id));963        if (remapped_name.empty()) {964            LLAMA_LOG_DEBUG("%s: pruning tensor %s\n", __func__, it.first.c_str());965            continue;966        }967 968        if (remapped_name != it.first) {969            ggml_set_name(it.second.tensor, remapped_name.c_str());970            LLAMA_LOG_DEBUG("%s: tensor %s remapped to %s\n", __func__, it.first.c_str(), ggml_get_name(it.second.tensor));971        }972        tensors.push_back(&it.second);973    }974    if (!prune_list.empty()) {975        gguf_set_val_u32(ctx_out.get(), ml.llm_kv(LLM_KV_BLOCK_COUNT).c_str(), blk_id);976    }977 978    // keep_split requires that the weights are sorted by split index979    if (params->keep_split) {980        std::sort(tensors.begin(), tensors.end(), [](const llama_model_loader::llama_tensor_weight * a, const llama_model_loader::llama_tensor_weight * b) {981            if (a->idx == b->idx) {982                return a->offs < b->offs;983            }984            return a->idx < b->idx;985        });986    }987 988    // compute tensor metadata once and cache it989    std::vector<tensor_metadata> metadata(tensors.size());990    for (size_t i = 0; i < tensors.size(); ++i) {991        metadata[i].name = ggml_get_name(tensors[i]->tensor);992    }993 994    // initialize quantization state counters and metadata categories995    init_quantize_state_counters(qs, metadata);996 997    int idx = 0;998    uint16_t n_split = 1;999 1000    // Assume split index is continuous1001    if (params->keep_split) {1002        for (const auto * it : tensors) {1003            n_split = std::max(uint16_t(it->idx + 1), n_split);1004        }1005    }1006    std::vector<gguf_context_ptr> ctx_outs(n_split);1007    ctx_outs[0] = std::move(ctx_out);1008 1009    // flag for --dry-run1010    bool will_require_imatrix = false;1011 1012    //1013    // preliminary iteration over all weights1014    //1015 1016    for (size_t i = 0; i < tensors.size(); ++i) {1017        const auto * it = tensors[i];1018        const struct ggml_tensor * tensor = it->tensor;1019 1020        uint16_t i_split = params->keep_split ? it->idx : 0;1021        if (!ctx_outs[i_split]) {1022            ctx_outs[i_split].reset(gguf_init_empty());1023        }1024        gguf_add_tensor(ctx_outs[i_split].get(), tensor);1025 1026        metadata[i].allows_quantization = tensor_allows_quantization(params, model.arch, tensor);1027 1028        if (metadata[i].allows_quantization) {1029            metadata[i].target_type = llama_tensor_get_type(qs, params, tensor, default_type, metadata[i]);1030        } else {1031            metadata[i].target_type = tensor->type;1032        }1033 1034        metadata[i].requires_imatrix = tensor_requires_imatrix(tensor->name, metadata[i].target_type, ftype);1035 1036        if (params->imatrix) {1037            metadata[i].remapped_imatrix_name = remap_imatrix(tensor->name, mapped);1038        } else if (metadata[i].allows_quantization && metadata[i].requires_imatrix) {1039            if (params->dry_run) {1040                will_require_imatrix = true;1041            } else {1042                LLAMA_LOG_ERROR("\n============================================================================\n"1043                                " ERROR: this quantization requires an importance matrix!\n"1044                                "        - offending tensor: %s\n"1045                                "        - target type: %s\n"1046                                "============================================================================\n\n",1047                                metadata[i].name.c_str(), ggml_type_name(metadata[i].target_type));1048                throw std::runtime_error("this quantization requires an imatrix!");1049            }1050        }1051    }1052 1053    // Set split info if needed1054    if (n_split > 1) {1055        for (size_t i = 0; i < ctx_outs.size(); ++i) {1056            gguf_set_val_u16(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_NO).c_str(), i);1057            gguf_set_val_u16(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_COUNT).c_str(), n_split);1058            gguf_set_val_i32(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_TENSORS_COUNT).c_str(), (int32_t)tensors.size());1059        }1060    }1061 1062    size_t total_size_org = 0;1063    size_t total_size_new = 0;1064 1065    std::vector<std::thread> workers;1066    workers.reserve(nthread);1067 1068    std::vector<no_init<uint8_t>> read_data;1069    std::vector<no_init<uint8_t>> work;1070    std::vector<no_init<float>> f32_conv_buf;1071 1072    int cur_split = -1;1073    std::ofstream fout;1074    auto close_ofstream = [&]() {1075        // Write metadata and close file handler1076        if (fout.is_open()) {1077            fout.seekp(0);1078            std::vector<uint8_t> data(gguf_get_meta_size(ctx_outs[cur_split].get()));1079            gguf_get_meta_data(ctx_outs[cur_split].get(), data.data());1080            fout.write((const char *) data.data(), data.size());1081            fout.close();1082        }1083    };1084    auto new_ofstream = [&](int index) {1085        cur_split = index;1086        GGML_ASSERT(ctx_outs[cur_split] && "Find uninitialized gguf_context");1087        std::string fname = fname_out;1088        if (params->keep_split) {1089            std::vector<char> split_path(llama_path_max(), 0);1090            llama_split_path(split_path.data(), split_path.size(), fname_out.c_str(), cur_split, n_split);1091            fname = std::string(split_path.data());1092        }1093 1094        fout = std::ofstream(fname, std::ios::binary);1095        fout.exceptions(std::ofstream::failbit); // fail fast on write errors1096        const size_t meta_size = gguf_get_meta_size(ctx_outs[cur_split].get());1097        // placeholder for the meta data1098        ::zeros(fout, meta_size);1099    };1100 1101    // no output file for --dry-run1102    if (!params->dry_run) {1103        new_ofstream(0);1104    }1105 1106    //1107    // main loop: iterate over all weights1108    //1109 1110    for (size_t i = 0; i < tensors.size(); ++i) {1111        const auto & weight = *tensors[i];1112        const auto & tm = metadata[i];1113        ggml_tensor * tensor = weight.tensor;1114 1115        if (!params->dry_run && (weight.idx != cur_split && params->keep_split)) {1116            close_ofstream();1117            new_ofstream(weight.idx);1118        }1119 1120        const size_t tensor_size = ggml_nbytes(tensor);1121 1122        if (!params->dry_run) {1123            if (!ml.use_mmap) {1124                if (read_data.size() < tensor_size) {1125                    read_data.resize(tensor_size);1126                }1127                tensor->data = read_data.data();1128            }1129            ml.load_data_for(tensor);1130        }1131 1132        LLAMA_LOG_INFO("[%4d/%4d] %-36s - [%s], type = %6s, ",1133               ++idx, ml.n_tensors,1134               ggml_get_name(tensor),1135               llama_format_tensor_shape(tensor).c_str(),1136               ggml_type_name(tensor->type));1137 1138        const ggml_type cur_type = tensor->type;1139        const ggml_type new_type = tm.target_type;1140 1141        // If we've decided to quantize to the same type the tensor is already1142        // in then there's nothing to do.1143        bool quantize = cur_type != new_type;1144 1145        void * new_data;1146        size_t new_size;1147 1148        if (params->dry_run) {1149            // the --dry-run option calculates the final quantization size without quantizing1150            if (quantize) {1151                new_size = ggml_nrows(tensor) * ggml_row_size(new_type, tensor->ne[0]);1152                LLAMA_LOG_INFO("size = %8.2f MiB -> %8.2f MiB (%s)\n",1153                               tensor_size/1024.0/1024.0,1154                               new_size/1024.0/1024.0,1155                               ggml_type_name(new_type));1156                if (!will_require_imatrix && tm.requires_imatrix) {1157                    will_require_imatrix = true;1158                }1159            } else {1160                new_size = tensor_size;1161                LLAMA_LOG_INFO("size = %8.3f MiB\n", new_size/1024.0/1024.0);1162            }1163            total_size_org += tensor_size;1164            total_size_new += new_size;1165            continue;1166        } else {1167            // no --dry-run, perform quantization1168            if (!quantize) {1169                new_data = tensor->data;1170                new_size = tensor_size;1171                LLAMA_LOG_INFO("size = %8.3f MiB\n", tensor_size/1024.0/1024.0);1172            } else {1173                const int64_t nelements = ggml_nelements(tensor);1174 1175                const float * imatrix = nullptr;1176                if (imatrix_data) {1177                    auto it = imatrix_data->find(tm.remapped_imatrix_name);1178                    if (it == imatrix_data->end()) {1179                        LLAMA_LOG_INFO("\n====== %s: did not find weights for %s\n", __func__, tensor->name);1180                    } else {1181                        if (it->second.size() == (size_t)tensor->ne[0]*tensor->ne[2]) {1182                            imatrix = it->second.data();1183                        } else {1184                            LLAMA_LOG_INFO("\n====== %s: imatrix size %d is different from tensor size %d for %s\n", __func__,1185                                    int(it->second.size()), int(tensor->ne[0]*tensor->ne[2]), tensor->name);1186 1187                            // this can happen when quantizing an old mixtral model with split tensors with a new incompatible imatrix1188                            // this is a significant error and it may be good idea to abort the process if this happens,1189                            // since many people will miss the error and not realize that most of the model is being quantized without an imatrix1190                            // tok_embd should be ignored in this case, since it always causes this warning1191                            if (!tensor_name_match_token_embd(tensor->name)) {1192                                throw std::runtime_error(format("imatrix size %d is different from tensor size %d for %s",1193                                        int(it->second.size()), int(tensor->ne[0]*tensor->ne[2]), tensor->name));1194                            }1195                        }1196                    }1197                }1198                if (!imatrix && tm.requires_imatrix) {1199                    LLAMA_LOG_ERROR("\n\n============================================================\n");1200                    LLAMA_LOG_ERROR("Missing importance matrix for tensor %s in a very low-bit quantization\n", tensor->name);

Showing the first 1,200 of 1403 lines. Download the file for the rest.