KBaba7/llama.cpp
0
1#include "llama-quant.h"2 3#include "llama-impl.h"4#include "llama-model.h"5#include "llama-model-loader.h"6 7#include <algorithm>8#include <cmath>9#include <cstring>10#include <cinttypes>11#include <fstream>12#include <mutex>13#include <thread>14#include <unordered_map>15 16static void zeros(std::ofstream & file, size_t n) {17 char zero = 0;18 for (size_t i = 0; i < n; ++i) {19 file.write(&zero, 1);20 }21}22 23struct quantize_state_impl {24 const llama_model & model;25 const llama_model_quantize_params * params;26 27 int n_attention_wv = 0;28 int n_ffn_down = 0;29 int n_ffn_gate = 0;30 int n_ffn_up = 0;31 int i_attention_wv = 0;32 int i_ffn_down = 0;33 int i_ffn_gate = 0;34 int i_ffn_up = 0;35 36 int n_k_quantized = 0;37 int n_fallback = 0;38 39 bool has_imatrix = false;40 41 // used to figure out if a model shares tok_embd with the output weight42 bool has_output = false;43 44 quantize_state_impl(const llama_model & model, const llama_model_quantize_params * params)45 : model(model)46 , params(params)47 {}48};49 50static void llama_tensor_dequantize_impl(51 struct ggml_tensor * tensor, std::vector<no_init<float>> & output, std::vector<std::thread> & workers,52 const size_t nelements, const int nthread53) {54 if (output.size() < nelements) {55 output.resize(nelements);56 }57 float * f32_output = (float *) output.data();58 59 const ggml_type_traits * qtype = ggml_get_type_traits(tensor->type);60 if (ggml_is_quantized(tensor->type)) {61 if (qtype->to_float == NULL) {62 throw std::runtime_error(format("type %s unsupported for integer quantization: no dequantization available", ggml_type_name(tensor->type)));63 }64 } else if (tensor->type != GGML_TYPE_F16 &&65 tensor->type != GGML_TYPE_BF16) {66 throw std::runtime_error(format("cannot dequantize/convert tensor type %s", ggml_type_name(tensor->type)));67 }68 69 if (nthread < 2) {70 if (tensor->type == GGML_TYPE_F16) {71 ggml_fp16_to_fp32_row((ggml_fp16_t *)tensor->data, f32_output, nelements);72 } else if (tensor->type == GGML_TYPE_BF16) {73 ggml_bf16_to_fp32_row((ggml_bf16_t *)tensor->data, f32_output, nelements);74 } else if (ggml_is_quantized(tensor->type)) {75 qtype->to_float(tensor->data, f32_output, nelements);76 } else {77 GGML_ABORT("fatal error"); // unreachable78 }79 return;80 }81 82 size_t block_size;83 if (tensor->type == GGML_TYPE_F16 ||84 tensor->type == GGML_TYPE_BF16) {85 block_size = 1;86 } else {87 block_size = (size_t)ggml_blck_size(tensor->type);88 }89 90 size_t block_size_bytes = ggml_type_size(tensor->type);91 92 GGML_ASSERT(nelements % block_size == 0);93 size_t nblocks = nelements / block_size;94 size_t blocks_per_thread = nblocks / nthread;95 size_t spare_blocks = nblocks - (blocks_per_thread * nthread); // if blocks aren't divisible by thread count96 97 size_t in_buff_offs = 0;98 size_t out_buff_offs = 0;99 100 for (int tnum = 0; tnum < nthread; tnum++) {101 size_t thr_blocks = blocks_per_thread + (tnum == nthread - 1 ? spare_blocks : 0); // num blocks for this thread102 size_t thr_elems = thr_blocks * block_size; // number of elements for this thread103 size_t thr_block_bytes = thr_blocks * block_size_bytes; // number of input bytes for this thread104 105 auto compute = [qtype] (ggml_type typ, uint8_t * inbuf, float * outbuf, int nels) {106 if (typ == GGML_TYPE_F16) {107 ggml_fp16_to_fp32_row((ggml_fp16_t *)inbuf, outbuf, nels);108 } else if (typ == GGML_TYPE_BF16) {109 ggml_bf16_to_fp32_row((ggml_bf16_t *)inbuf, outbuf, nels);110 } else {111 qtype->to_float(inbuf, outbuf, nels);112 }113 };114 workers.emplace_back(compute, tensor->type, (uint8_t *) tensor->data + in_buff_offs, f32_output + out_buff_offs, thr_elems);115 in_buff_offs += thr_block_bytes;116 out_buff_offs += thr_elems;117 }118 for (auto & w : workers) { w.join(); }119 workers.clear();120}121 122static ggml_type llama_tensor_get_type(quantize_state_impl & qs, ggml_type new_type, const ggml_tensor * tensor, llama_ftype ftype) {123 const std::string name = ggml_get_name(tensor);124 125 // TODO: avoid hardcoded tensor names - use the TN_* constants126 const llm_arch arch = qs.model.arch;127 const auto tn = LLM_TN(arch);128 129 auto use_more_bits = [](int i_layer, int n_layers) -> bool {130 return i_layer < n_layers/8 || i_layer >= 7*n_layers/8 || (i_layer - n_layers/8)%3 == 2;131 };132 const int n_expert = std::max(1, (int)qs.model.hparams.n_expert);133 auto layer_info = [n_expert] (int i_layer, int n_layer, const char * name) {134 if (n_expert > 1) {135 // Believe it or not, "experts" in the FFN of Mixtral-8x7B are not consecutive, but occasionally randomly136 // sprinkled in the model. Hence, simply dividing i_ffn_down by n_expert does not work137 // for getting the current layer as I initially thought, and we need to resort to parsing the138 // tensor name.139 if (sscanf(name, "blk.%d.", &i_layer) != 1) {140 throw std::runtime_error(format("Failed to determine layer for tensor %s", name));141 }142 if (i_layer < 0 || i_layer >= n_layer) {143 throw std::runtime_error(format("Bad layer %d for tensor %s. Must be in [0, %d)", i_layer, name, n_layer));144 }145 }146 return std::make_pair(i_layer, n_layer);147 };148 149 // for arches that share the same tensor between the token embeddings and the output, we quantize the token embeddings150 // with the quantization of the output tensor151 if (name == tn(LLM_TENSOR_OUTPUT, "weight") || (!qs.has_output && name == tn(LLM_TENSOR_TOKEN_EMBD, "weight"))) {152 if (qs.params->output_tensor_type < GGML_TYPE_COUNT) {153 new_type = qs.params->output_tensor_type;154 } else {155 const int64_t nx = tensor->ne[0];156 const int64_t qk_k = ggml_blck_size(new_type);157 158 if (arch == LLM_ARCH_FALCON || nx % qk_k != 0) {159 new_type = GGML_TYPE_Q8_0;160 }161 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS ||162 ftype == LLAMA_FTYPE_MOSTLY_IQ1_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ||163 ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {164 new_type = GGML_TYPE_Q5_K;165 }166 else if (new_type != GGML_TYPE_Q8_0) {167 new_type = GGML_TYPE_Q6_K;168 }169 }170 } else if (name == "token_embd.weight") {171 if (qs.params->token_embedding_type < GGML_TYPE_COUNT) {172 new_type = qs.params->token_embedding_type;173 } else {174 if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS ||175 ftype == LLAMA_FTYPE_MOSTLY_IQ1_S || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {176 new_type = GGML_TYPE_Q2_K;177 }178 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M) {179 new_type = GGML_TYPE_IQ3_S;180 }181 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {182 new_type = GGML_TYPE_IQ3_S;183 }184 else if (ftype == LLAMA_FTYPE_MOSTLY_TQ1_0 || ftype == LLAMA_FTYPE_MOSTLY_TQ2_0) {185 new_type = GGML_TYPE_Q4_K;186 }187 }188 } else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS || ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ1_S ||189 ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) {190 if (name.find("attn_v.weight") != std::string::npos) {191 if (qs.model.hparams.n_gqa() >= 4 || qs.model.hparams.n_expert >= 4) new_type = GGML_TYPE_Q4_K;192 else new_type = ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ? GGML_TYPE_IQ3_S : GGML_TYPE_Q2_K;193 ++qs.i_attention_wv;194 }195 else if (qs.model.hparams.n_expert == 8 && name.find("attn_k.weight") != std::string::npos) {196 new_type = GGML_TYPE_Q4_K;197 }198 else if (name.find("ffn_down") != std::string::npos) {199 if (qs.i_ffn_down < qs.n_ffn_down/8) {200 new_type = ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M ? GGML_TYPE_IQ3_S : GGML_TYPE_Q2_K;201 }202 ++qs.i_ffn_down;203 }204 else if (name.find("attn_output.weight") != std::string::npos) {205 if (qs.model.hparams.n_expert == 8) {206 new_type = GGML_TYPE_Q5_K;207 } else {208 if (ftype == LLAMA_FTYPE_MOSTLY_IQ1_S || ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) new_type = GGML_TYPE_IQ2_XXS;209 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ2_S || ftype == LLAMA_FTYPE_MOSTLY_IQ2_M) new_type = GGML_TYPE_IQ3_S;210 }211 }212 } else if (name.find("attn_v.weight") != std::string::npos) {213 if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) {214 new_type = qs.model.hparams.n_gqa() >= 4 ? GGML_TYPE_Q4_K : GGML_TYPE_Q3_K;215 }216 else if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S && qs.model.hparams.n_gqa() >= 4) {217 new_type = GGML_TYPE_Q4_K;218 }219 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {220 new_type = qs.model.hparams.n_gqa() >= 4 ? GGML_TYPE_Q4_K : !qs.has_imatrix ? GGML_TYPE_IQ3_S : GGML_TYPE_IQ3_XXS;221 }222 else if ((ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_S) && qs.model.hparams.n_gqa() >= 4) {223 new_type = GGML_TYPE_Q4_K;224 }225 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M) {226 new_type = GGML_TYPE_Q4_K;227 }228 else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {229 new_type = qs.i_attention_wv < 2 ? GGML_TYPE_Q5_K : GGML_TYPE_Q4_K;230 }231 else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) new_type = GGML_TYPE_Q5_K;232 else if ((ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) && qs.model.hparams.n_gqa() >= 4) {233 new_type = GGML_TYPE_Q5_K;234 }235 else if ((ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) &&236 use_more_bits(qs.i_attention_wv, qs.n_attention_wv)) new_type = GGML_TYPE_Q6_K;237 else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S && qs.i_attention_wv < 4) new_type = GGML_TYPE_Q5_K;238 if (qs.model.type == LLM_TYPE_70B) {239 // In the 70B model we have 8 heads sharing the same attn_v weights. As a result, the attn_v.weight tensor is240 // 8x smaller compared to attn_q.weight. Hence, we can get a nice boost in quantization accuracy with241 // nearly negligible increase in model size by quantizing this tensor with more bits:242 if (new_type == GGML_TYPE_Q3_K || new_type == GGML_TYPE_Q4_K) new_type = GGML_TYPE_Q5_K;243 }244 if (qs.model.hparams.n_expert == 8) {245 // for the 8-expert model, bumping this to Q8_0 trades just ~128MB246 // TODO: explore better strategies247 new_type = GGML_TYPE_Q8_0;248 }249 ++qs.i_attention_wv;250 } else if (name.find("attn_k.weight") != std::string::npos) {251 if (qs.model.hparams.n_expert == 8) {252 // for the 8-expert model, bumping this to Q8_0 trades just ~128MB253 // TODO: explore better strategies254 new_type = GGML_TYPE_Q8_0;255 }256 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS) {257 new_type = GGML_TYPE_IQ3_XXS;258 }259 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {260 new_type = GGML_TYPE_IQ2_S;261 }262 } else if (name.find("attn_q.weight") != std::string::npos) {263 if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS) {264 new_type = GGML_TYPE_IQ3_XXS;265 }266 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) {267 new_type = GGML_TYPE_IQ2_S;268 }269 } else if (name.find("ffn_down") != std::string::npos) {270 auto info = layer_info(qs.i_ffn_down, qs.n_ffn_down, name.c_str());271 int i_layer = info.first, n_layer = info.second;272 if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) new_type = GGML_TYPE_Q3_K;273 else if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S) {274 if (i_layer < n_layer/8) new_type = GGML_TYPE_Q4_K;275 }276 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS && !qs.has_imatrix) {277 new_type = i_layer < n_layer/8 ? GGML_TYPE_Q4_K : GGML_TYPE_Q3_K;278 }279 else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M) {280 new_type = i_layer < n_layer/16 ? GGML_TYPE_Q5_K281 : arch != LLM_ARCH_FALCON || use_more_bits(i_layer, n_layer) ? GGML_TYPE_Q4_K282 : GGML_TYPE_Q3_K;283 }284 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M && (i_layer < n_layer/8 ||285 (qs.model.hparams.n_expert == 8 && use_more_bits(i_layer, n_layer)))) {286 new_type = GGML_TYPE_Q4_K;287 }288 else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) {289 new_type = arch == LLM_ARCH_FALCON ? GGML_TYPE_Q4_K : GGML_TYPE_Q5_K;290 }291 else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) {292 if (arch == LLM_ARCH_FALCON) {293 new_type = i_layer < n_layer/16 ? GGML_TYPE_Q6_K :294 use_more_bits(i_layer, n_layer) ? GGML_TYPE_Q5_K : GGML_TYPE_Q4_K;295 } else {296 if (use_more_bits(i_layer, n_layer)) new_type = GGML_TYPE_Q6_K;297 }298 }299 else if (i_layer < n_layer/8 && (ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) && !qs.has_imatrix) {300 new_type = GGML_TYPE_Q5_K;301 }302 else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M && use_more_bits(i_layer, n_layer)) new_type = GGML_TYPE_Q6_K;303 else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S && arch != LLM_ARCH_FALCON && i_layer < n_layer/8) {304 new_type = GGML_TYPE_Q5_K;305 }306 else if ((ftype == LLAMA_FTYPE_MOSTLY_Q4_0 || ftype == LLAMA_FTYPE_MOSTLY_Q5_0)307 && qs.has_imatrix && i_layer < n_layer/8) {308 // Guard against craziness in the first few ffn_down layers that can happen even with imatrix for Q4_0/Q5_0.309 // We only do it when an imatrix is provided because a) we want to make sure that one can always get the310 // same quantization as before imatrix stuff, and b) Q4_1/Q5_1 do go crazy on ffn_down without an imatrix.311 new_type = ftype == LLAMA_FTYPE_MOSTLY_Q4_0 ? GGML_TYPE_Q4_1 : GGML_TYPE_Q5_1;312 }313 ++qs.i_ffn_down;314 } else if (name.find("attn_output.weight") != std::string::npos) {315 if (arch != LLM_ARCH_FALCON) {316 if (qs.model.hparams.n_expert == 8) {317 if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS || ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS ||318 ftype == LLAMA_FTYPE_MOSTLY_Q3_K_S || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M || ftype == LLAMA_FTYPE_MOSTLY_IQ4_NL ||319 ftype == LLAMA_FTYPE_MOSTLY_Q4_K_S || ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M || ftype == LLAMA_FTYPE_MOSTLY_IQ3_S ||320 ftype == LLAMA_FTYPE_MOSTLY_IQ3_M || ftype == LLAMA_FTYPE_MOSTLY_IQ4_XS) {321 new_type = GGML_TYPE_Q5_K;322 }323 } else {324 if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K ) new_type = GGML_TYPE_Q3_K;325 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XXS) new_type = GGML_TYPE_IQ3_S;326 else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M ) new_type = GGML_TYPE_Q4_K;327 else if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L ) new_type = GGML_TYPE_Q5_K;328 else if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_M ) new_type = GGML_TYPE_Q4_K;329 }330 } else {331 if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L) new_type = GGML_TYPE_Q4_K;332 }333 }334 else if (name.find("attn_qkv.weight") != std::string::npos) {335 if (ftype == LLAMA_FTYPE_MOSTLY_Q3_K_M || ftype == LLAMA_FTYPE_MOSTLY_Q3_K_L || ftype == LLAMA_FTYPE_MOSTLY_IQ3_M) {336 new_type = GGML_TYPE_Q4_K;337 }338 else if (ftype == LLAMA_FTYPE_MOSTLY_Q4_K_M) new_type = GGML_TYPE_Q5_K;339 else if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_M) new_type = GGML_TYPE_Q6_K;340 }341 else if (name.find("ffn_gate") != std::string::npos) {342 auto info = layer_info(qs.i_ffn_gate, qs.n_ffn_gate, name.c_str());343 int i_layer = info.first, n_layer = info.second;344 if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS && (i_layer >= n_layer/8 && i_layer < 7*n_layer/8)) {345 new_type = GGML_TYPE_IQ3_XXS;346 }347 ++qs.i_ffn_gate;348 }349 else if (name.find("ffn_up") != std::string::npos) {350 auto info = layer_info(qs.i_ffn_up, qs.n_ffn_up, name.c_str());351 int i_layer = info.first, n_layer = info.second;352 if (ftype == LLAMA_FTYPE_MOSTLY_IQ3_XS && (i_layer >= n_layer/8 && i_layer < 7*n_layer/8)) {353 new_type = GGML_TYPE_IQ3_XXS;354 }355 ++qs.i_ffn_up;356 }357 358 // if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) new_type = GGML_TYPE_Q3_K;359 //}360 // IK: let's remove this, else Q2_K is almost the same as Q3_K_S361 //else if (name.find("ffn_gate") != std::string::npos || name.find("ffn_up") != std::string::npos) {362 // if (ftype == LLAMA_FTYPE_MOSTLY_Q2_K) new_type = GGML_TYPE_Q3_K;363 //}364 // This can be used to reduce the size of the Q5_K_S model.365 // The associated PPL increase is fully in line with the size reduction366 //else {367 // if (ftype == LLAMA_FTYPE_MOSTLY_Q5_K_S) new_type = GGML_TYPE_Q4_K;368 //}369 bool convert_incompatible_tensor = false;370 {371 const int64_t nx = tensor->ne[0];372 const int64_t ny = tensor->ne[1];373 const int64_t qk_k = ggml_blck_size(new_type);374 375 if (nx % qk_k != 0) {376 LLAMA_LOG_WARN("\n\n%s : tensor cols %" PRId64 " x %" PRId64 " are not divisible by %" PRId64 ", required for %s", __func__, nx, ny, qk_k, ggml_type_name(new_type));377 convert_incompatible_tensor = true;378 } else {379 ++qs.n_k_quantized;380 }381 }382 383 if (convert_incompatible_tensor) {384 switch (new_type) {385 case GGML_TYPE_TQ1_0:386 case GGML_TYPE_TQ2_0: new_type = GGML_TYPE_Q4_0; break; // TODO: use a symmetric type instead387 case GGML_TYPE_IQ2_XXS:388 case GGML_TYPE_IQ2_XS:389 case GGML_TYPE_IQ2_S:390 case GGML_TYPE_IQ3_XXS:391 case GGML_TYPE_IQ3_S:392 case GGML_TYPE_IQ1_S:393 case GGML_TYPE_IQ1_M:394 case GGML_TYPE_Q2_K:395 case GGML_TYPE_Q3_K:396 case GGML_TYPE_IQ4_XS: new_type = GGML_TYPE_IQ4_NL; break;397 case GGML_TYPE_Q4_K: new_type = GGML_TYPE_Q5_0; break;398 case GGML_TYPE_Q5_K: new_type = GGML_TYPE_Q5_1; break;399 case GGML_TYPE_Q6_K: new_type = GGML_TYPE_Q8_0; break;400 default: throw std::runtime_error("\nUnsupported tensor size encountered\n");401 }402 if (tensor->ne[0] % ggml_blck_size(new_type) != 0) {403 new_type = GGML_TYPE_F16;404 }405 LLAMA_LOG_WARN(" - using fallback quantization %s\n", ggml_type_name(new_type));406 ++qs.n_fallback;407 }408 409 return new_type;410}411 412static size_t llama_tensor_quantize_impl(enum ggml_type new_type, const float * f32_data, void * new_data, const int64_t chunk_size, int64_t nrows, int64_t n_per_row, const float * imatrix, std::vector<std::thread> & workers, const int nthread) {413 if (nthread < 2) {414 // single-thread415 size_t new_size = ggml_quantize_chunk(new_type, f32_data, new_data, 0, nrows, n_per_row, imatrix);416 if (!ggml_validate_row_data(new_type, new_data, new_size)) {417 throw std::runtime_error("quantized data validation failed");418 }419 return new_size;420 }421 422 std::mutex mutex;423 int64_t counter = 0;424 size_t new_size = 0;425 bool valid = true;426 auto compute = [&mutex, &counter, &new_size, &valid, new_type, f32_data, new_data, chunk_size,427 nrows, n_per_row, imatrix]() {428 const int64_t nrows_per_chunk = chunk_size / n_per_row;429 size_t local_size = 0;430 while (true) {431 std::unique_lock<std::mutex> lock(mutex);432 int64_t first_row = counter; counter += nrows_per_chunk;433 if (first_row >= nrows) {434 if (local_size > 0) {435 new_size += local_size;436 }437 break;438 }439 lock.unlock();440 const int64_t this_nrow = std::min(nrows - first_row, nrows_per_chunk);441 size_t this_size = ggml_quantize_chunk(new_type, f32_data, new_data, first_row * n_per_row, this_nrow, n_per_row, imatrix);442 local_size += this_size;443 444 // validate the quantized data445 const size_t row_size = ggml_row_size(new_type, n_per_row);446 void * this_data = (char *) new_data + first_row * row_size;447 if (!ggml_validate_row_data(new_type, this_data, this_size)) {448 std::unique_lock<std::mutex> lock(mutex);449 valid = false;450 break;451 }452 }453 };454 for (int it = 0; it < nthread - 1; ++it) {455 workers.emplace_back(compute);456 }457 compute();458 for (auto & w : workers) { w.join(); }459 workers.clear();460 if (!valid) {461 throw std::runtime_error("quantized data validation failed");462 }463 return new_size;464}465 466static void llama_model_quantize_impl(const std::string & fname_inp, const std::string & fname_out, const llama_model_quantize_params * params) {467 ggml_type default_type;468 llama_ftype ftype = params->ftype;469 470 switch (params->ftype) {471 case LLAMA_FTYPE_MOSTLY_Q4_0: default_type = GGML_TYPE_Q4_0; break;472 case LLAMA_FTYPE_MOSTLY_Q4_1: default_type = GGML_TYPE_Q4_1; break;473 case LLAMA_FTYPE_MOSTLY_Q5_0: default_type = GGML_TYPE_Q5_0; break;474 case LLAMA_FTYPE_MOSTLY_Q5_1: default_type = GGML_TYPE_Q5_1; break;475 case LLAMA_FTYPE_MOSTLY_Q8_0: default_type = GGML_TYPE_Q8_0; break;476 case LLAMA_FTYPE_MOSTLY_F16: default_type = GGML_TYPE_F16; break;477 case LLAMA_FTYPE_MOSTLY_BF16: default_type = GGML_TYPE_BF16; break;478 case LLAMA_FTYPE_ALL_F32: default_type = GGML_TYPE_F32; break;479 480 // K-quants481 case LLAMA_FTYPE_MOSTLY_Q2_K_S:482 case LLAMA_FTYPE_MOSTLY_Q2_K: default_type = GGML_TYPE_Q2_K; break;483 case LLAMA_FTYPE_MOSTLY_IQ3_XS: default_type = GGML_TYPE_IQ3_S; break;484 case LLAMA_FTYPE_MOSTLY_Q3_K_S:485 case LLAMA_FTYPE_MOSTLY_Q3_K_M:486 case LLAMA_FTYPE_MOSTLY_Q3_K_L: default_type = GGML_TYPE_Q3_K; break;487 case LLAMA_FTYPE_MOSTLY_Q4_K_S:488 case LLAMA_FTYPE_MOSTLY_Q4_K_M: default_type = GGML_TYPE_Q4_K; break;489 case LLAMA_FTYPE_MOSTLY_Q5_K_S:490 case LLAMA_FTYPE_MOSTLY_Q5_K_M: default_type = GGML_TYPE_Q5_K; break;491 case LLAMA_FTYPE_MOSTLY_Q6_K: default_type = GGML_TYPE_Q6_K; break;492 case LLAMA_FTYPE_MOSTLY_TQ1_0: default_type = GGML_TYPE_TQ1_0; break;493 case LLAMA_FTYPE_MOSTLY_TQ2_0: default_type = GGML_TYPE_TQ2_0; break;494 case LLAMA_FTYPE_MOSTLY_IQ2_XXS: default_type = GGML_TYPE_IQ2_XXS; break;495 case LLAMA_FTYPE_MOSTLY_IQ2_XS: default_type = GGML_TYPE_IQ2_XS; break;496 case LLAMA_FTYPE_MOSTLY_IQ2_S: default_type = GGML_TYPE_IQ2_XS; break;497 case LLAMA_FTYPE_MOSTLY_IQ2_M: default_type = GGML_TYPE_IQ2_S; break;498 case LLAMA_FTYPE_MOSTLY_IQ3_XXS: default_type = GGML_TYPE_IQ3_XXS; break;499 case LLAMA_FTYPE_MOSTLY_IQ1_S: default_type = GGML_TYPE_IQ1_S; break;500 case LLAMA_FTYPE_MOSTLY_IQ1_M: default_type = GGML_TYPE_IQ1_M; break;501 case LLAMA_FTYPE_MOSTLY_IQ4_NL: default_type = GGML_TYPE_IQ4_NL; break;502 case LLAMA_FTYPE_MOSTLY_IQ4_XS: default_type = GGML_TYPE_IQ4_XS; break;503 case LLAMA_FTYPE_MOSTLY_IQ3_S: default_type = GGML_TYPE_IQ3_S; break;504 case LLAMA_FTYPE_MOSTLY_IQ3_M: default_type = GGML_TYPE_IQ3_S; break;505 506 default: throw std::runtime_error(format("invalid output file type %d\n", ftype));507 }508 509 int nthread = params->nthread;510 511 if (nthread <= 0) {512 nthread = std::thread::hardware_concurrency();513 }514 515 // mmap consistently increases speed Linux, and also increases speed on Windows with516 // hot cache. It may cause a slowdown on macOS, possibly related to free memory.517#if defined(__linux__) || defined(_WIN32)518 constexpr bool use_mmap = true;519#else520 constexpr bool use_mmap = false;521#endif522 523 llama_model_kv_override * kv_overrides = nullptr;524 if (params->kv_overrides) {525 auto v = (std::vector<llama_model_kv_override>*)params->kv_overrides;526 kv_overrides = v->data();527 }528 529 std::vector<std::string> splits = {};530 llama_model_loader ml(fname_inp, splits, use_mmap, /*check_tensors*/ true, kv_overrides);531 ml.init_mappings(false); // no prefetching532 533 llama_model model(llama_model_default_params());534 535 model.load_arch (ml);536 model.load_hparams(ml);537 model.load_stats (ml);538 539 struct quantize_state_impl qs(model, params);540 541 if (params->only_copy) {542 ftype = ml.ftype;543 }544 const std::unordered_map<std::string, std::vector<float>> * imatrix_data = nullptr;545 if (params->imatrix) {546 imatrix_data = static_cast<const std::unordered_map<std::string, std::vector<float>>*>(params->imatrix);547 if (imatrix_data) {548 LLAMA_LOG_INFO("================================ Have weights data with %d entries\n",int(imatrix_data->size()));549 qs.has_imatrix = true;550 // check imatrix for nans or infs551 for (const auto & kv : *imatrix_data) {552 for (float f : kv.second) {553 if (!std::isfinite(f)) {554 throw std::runtime_error(format("imatrix contains non-finite value %f\n", f));555 }556 }557 }558 }559 }560 561 const size_t align = GGUF_DEFAULT_ALIGNMENT;562 gguf_context_ptr ctx_out { gguf_init_empty() };563 564 // copy the KV pairs from the input file565 gguf_set_kv (ctx_out.get(), ml.meta.get());566 gguf_set_val_u32(ctx_out.get(), "general.quantization_version", GGML_QNT_VERSION); // TODO: use LLM_KV567 gguf_set_val_u32(ctx_out.get(), "general.file_type", ftype); // TODO: use LLM_KV568 569 // Remove split metadata570 gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_NO).c_str());571 gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_COUNT).c_str());572 gguf_remove_key(ctx_out.get(), ml.llm_kv(LLM_KV_SPLIT_TENSORS_COUNT).c_str());573 574 if (params->kv_overrides) {575 const std::vector<llama_model_kv_override> & overrides = *(const std::vector<llama_model_kv_override> *)params->kv_overrides;576 for (const auto & o : overrides) {577 if (o.key[0] == 0) break;578 if (o.tag == LLAMA_KV_OVERRIDE_TYPE_FLOAT) {579 gguf_set_val_f32(ctx_out.get(), o.key, o.val_f64);580 } else if (o.tag == LLAMA_KV_OVERRIDE_TYPE_INT) {581 gguf_set_val_i32(ctx_out.get(), o.key, o.val_i64);582 } else if (o.tag == LLAMA_KV_OVERRIDE_TYPE_BOOL) {583 gguf_set_val_bool(ctx_out.get(), o.key, o.val_bool);584 } else if (o.tag == LLAMA_KV_OVERRIDE_TYPE_STR) {585 gguf_set_val_str(ctx_out.get(), o.key, o.val_str);586 } else {587 LLAMA_LOG_WARN("%s: unknown KV override type for key %s\n", __func__, o.key);588 }589 }590 }591 592 // make a list of weights593 std::vector<const llama_model_loader::llama_tensor_weight *> tensors;594 tensors.reserve(ml.weights_map.size());595 for (const auto & it : ml.weights_map) {596 tensors.push_back(&it.second);597 }598 599 // keep_split requires that the weights are sorted by split index600 if (params->keep_split) {601 std::sort(tensors.begin(), tensors.end(), [](const llama_model_loader::llama_tensor_weight * a, const llama_model_loader::llama_tensor_weight * b) {602 if (a->idx == b->idx) {603 return a->offs < b->offs;604 }605 return a->idx < b->idx;606 });607 }608 609 for (const auto * it : tensors) {610 const struct ggml_tensor * tensor = it->tensor;611 612 const std::string name = ggml_get_name(tensor);613 614 // TODO: avoid hardcoded tensor names - use the TN_* constants615 if (name.find("attn_v.weight") != std::string::npos ||616 name.find("attn_qkv.weight") != std::string::npos ||617 name.find("attn_kv_b.weight")!= std::string::npos) {618 ++qs.n_attention_wv;619 } else if (name == LLM_TN(model.arch)(LLM_TENSOR_OUTPUT, "weight")) {620 qs.has_output = true;621 }622 }623 624 qs.n_ffn_down = qs.n_ffn_gate = qs.n_ffn_up = (int)model.hparams.n_layer;625 626 // sanity checks for models that have attention layers627 if (qs.n_attention_wv != 0)628 {629 const auto & n_head_kv_iter = model.hparams.n_head_kv_arr.begin();630 // attention layers have a non-zero number of kv heads631 int32_t n_attn_layer = model.hparams.n_layer - std::count(n_head_kv_iter, n_head_kv_iter + model.hparams.n_layer, 0);632 if (llama_model_has_encoder(&model)) {633 n_attn_layer *= 3;634 }635 GGML_ASSERT((qs.n_attention_wv == n_attn_layer) && "n_attention_wv is unexpected");636 }637 638 size_t total_size_org = 0;639 size_t total_size_new = 0;640 641 std::vector<std::thread> workers;642 workers.reserve(nthread);643 644 int idx = 0;645 646 std::vector<no_init<uint8_t>> read_data;647 std::vector<no_init<uint8_t>> work;648 std::vector<no_init<float>> f32_conv_buf;649 650 uint16_t n_split = 1;651 652 // Assume split index is continuous653 if (params->keep_split) {654 for (const auto * it : tensors) {655 n_split = std::max(uint16_t(it->idx + 1), n_split);656 }657 }658 std::vector<gguf_context_ptr> ctx_outs(n_split);659 ctx_outs[0] = std::move(ctx_out);660 661 // populate the original tensors so we get an initial meta data662 for (const auto * it : tensors) {663 uint16_t i_split = params->keep_split ? it->idx : 0;664 struct ggml_tensor * tensor = it->tensor;665 if (!ctx_outs[i_split]) {666 ctx_outs[i_split].reset(gguf_init_empty());667 }668 gguf_add_tensor(ctx_outs[i_split].get(), tensor);669 }670 671 // Set split info if needed672 if (n_split > 1) {673 for (size_t i = 0; i < ctx_outs.size(); ++i) {674 gguf_set_val_u16(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_NO).c_str(), i);675 gguf_set_val_u16(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_COUNT).c_str(), n_split);676 gguf_set_val_i32(ctx_outs[i].get(), ml.llm_kv(LLM_KV_SPLIT_TENSORS_COUNT).c_str(), ml.n_tensors);677 }678 }679 680 int cur_split = -1;681 std::ofstream fout;682 auto close_ofstream = [&]() {683 // Write metadata and close file handler684 if (fout.is_open()) {685 fout.seekp(0);686 std::vector<uint8_t> data(gguf_get_meta_size(ctx_outs[cur_split].get()));687 gguf_get_meta_data(ctx_outs[cur_split].get(), data.data());688 fout.write((const char *) data.data(), data.size());689 fout.close();690 }691 };692 auto new_ofstream = [&](int index) {693 cur_split = index;694 GGML_ASSERT(ctx_outs[cur_split] && "Find uninitialized gguf_context");695 std::string fname = fname_out;696 if (params->keep_split) {697 std::vector<char> split_path(llama_path_max(), 0);698 llama_split_path(split_path.data(), split_path.size(), fname_out.c_str(), cur_split, n_split);699 fname = std::string(split_path.data());700 }701 702 fout = std::ofstream(fname, std::ios::binary);703 fout.exceptions(std::ofstream::failbit); // fail fast on write errors704 const size_t meta_size = gguf_get_meta_size(ctx_outs[cur_split].get());705 // placeholder for the meta data706 ::zeros(fout, meta_size);707 };708 709 const auto tn = LLM_TN(model.arch);710 new_ofstream(0);711 for (const auto * it : tensors) {712 const auto & weight = *it;713 struct ggml_tensor * tensor = weight.tensor;714 if (weight.idx != cur_split && params->keep_split) {715 close_ofstream();716 new_ofstream(weight.idx);717 }718 719 const std::string name = ggml_get_name(tensor);720 721 if (!ml.use_mmap) {722 if (read_data.size() < ggml_nbytes(tensor)) {723 read_data.resize(ggml_nbytes(tensor));724 }725 tensor->data = read_data.data();726 }727 ml.load_data_for(tensor);728 729 LLAMA_LOG_INFO("[%4d/%4d] %36s - [%s], type = %6s, ",730 ++idx, ml.n_tensors,731 ggml_get_name(tensor),732 llama_format_tensor_shape(tensor).c_str(),733 ggml_type_name(tensor->type));734 735 // This used to be a regex, but <regex> has an extreme cost to compile times.736 bool quantize = name.rfind("weight") == name.size() - 6; // ends with 'weight'?737 738 // quantize only 2D and 3D tensors (experts)739 quantize &= (ggml_n_dims(tensor) >= 2);740 741 // do not quantize norm tensors742 quantize &= name.find("_norm.weight") == std::string::npos;743 744 quantize &= params->quantize_output_tensor || name != "output.weight";745 quantize &= !params->only_copy;746 747 // do not quantize expert gating tensors748 // NOTE: can't use LLM_TN here because the layer number is not known749 quantize &= name.find("ffn_gate_inp.weight") == std::string::npos;750 751 // do not quantize positional embeddings and token types (BERT)752 quantize &= name != LLM_TN(model.arch)(LLM_TENSOR_POS_EMBD, "weight");753 quantize &= name != LLM_TN(model.arch)(LLM_TENSOR_TOKEN_TYPES, "weight");754 755 // do not quantize Mamba's small yet 2D weights756 // NOTE: can't use LLM_TN here because the layer number is not known757 quantize &= name.find("ssm_conv1d.weight") == std::string::npos;758 759 // do not quantize RWKV's time_mix_first tensors760 quantize &= name.find("time_mix_first.weight") == std::string::npos;761 quantize &= name.find("time_mix_w1.weight") == std::string::npos;762 quantize &= name.find("time_mix_w2.weight") == std::string::npos;763 quantize &= name.find("time_mix_decay_w1.weight") == std::string::npos;764 quantize &= name.find("time_mix_decay_w2.weight") == std::string::npos;765 quantize &= name.find("time_mix_lerp_fused.weight") == std::string::npos;766 767 // do not quantize relative position bias (T5)768 quantize &= name.find("attn_rel_b.weight") == std::string::npos;769 770 enum ggml_type new_type;771 void * new_data;772 size_t new_size;773 774 if (quantize) {775 new_type = default_type;776 777 // get more optimal quantization type based on the tensor shape, layer, etc.778 if (!params->pure && ggml_is_quantized(default_type)) {779 new_type = llama_tensor_get_type(qs, new_type, tensor, ftype);780 }781 if (params->token_embedding_type < GGML_TYPE_COUNT && strcmp(tensor->name, "token_embd.weight") == 0) {782 new_type = params->token_embedding_type;783 }784 if (params->output_tensor_type < GGML_TYPE_COUNT && strcmp(tensor->name, "output.weight") == 0) {785 new_type = params->output_tensor_type;786 }787 788 // If we've decided to quantize to the same type the tensor is already789 // in then there's nothing to do.790 quantize = tensor->type != new_type;791 }792 793 if (!quantize) {794 new_type = tensor->type;795 new_data = tensor->data;796 new_size = ggml_nbytes(tensor);797 LLAMA_LOG_INFO("size = %8.3f MB\n", ggml_nbytes(tensor)/1024.0/1024.0);798 } else {799 const int64_t nelements = ggml_nelements(tensor);800 801 const float * imatrix = nullptr;802 if (imatrix_data) {803 auto it = imatrix_data->find(tensor->name);804 if (it == imatrix_data->end()) {805 LLAMA_LOG_INFO("\n====== %s: did not find weights for %s\n", __func__, tensor->name);806 } else {807 if (it->second.size() == (size_t)tensor->ne[0]*tensor->ne[2]) {808 imatrix = it->second.data();809 } else {810 LLAMA_LOG_INFO("\n====== %s: imatrix size %d is different from tensor size %d for %s\n", __func__,811 int(it->second.size()), int(tensor->ne[0]*tensor->ne[2]), tensor->name);812 813 // this can happen when quantizing an old mixtral model with split tensors with a new incompatible imatrix814 // this is a significant error and it may be good idea to abort the process if this happens,815 // since many people will miss the error and not realize that most of the model is being quantized without an imatrix816 // tok_embd should be ignored in this case, since it always causes this warning817 if (name != tn(LLM_TENSOR_TOKEN_EMBD, "weight")) {818 throw std::runtime_error(format("imatrix size %d is different from tensor size %d for %s",819 int(it->second.size()), int(tensor->ne[0]*tensor->ne[2]), tensor->name));820 }821 }822 }823 }824 if ((new_type == GGML_TYPE_IQ2_XXS ||825 new_type == GGML_TYPE_IQ2_XS ||826 new_type == GGML_TYPE_IQ2_S ||827 new_type == GGML_TYPE_IQ1_S ||828 (new_type == GGML_TYPE_IQ1_M && strcmp(tensor->name, "token_embd.weight") && strcmp(tensor->name, "output.weight")) ||829 (new_type == GGML_TYPE_Q2_K && params->ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S && strcmp(tensor->name, "token_embd.weight") != 0)) && !imatrix) {830 LLAMA_LOG_ERROR("\n\n============================================================\n");831 LLAMA_LOG_ERROR("Missing importance matrix for tensor %s in a very low-bit quantization\n", tensor->name);832 LLAMA_LOG_ERROR("The result will be garbage, so bailing out\n");833 LLAMA_LOG_ERROR("============================================================\n\n");834 throw std::runtime_error(format("Missing importance matrix for tensor %s in a very low-bit quantization", tensor->name));835 }836 837 float * f32_data;838 839 if (tensor->type == GGML_TYPE_F32) {840 f32_data = (float *) tensor->data;841 } else if (ggml_is_quantized(tensor->type) && !params->allow_requantize) {842 throw std::runtime_error(format("requantizing from type %s is disabled", ggml_type_name(tensor->type)));843 } else {844 llama_tensor_dequantize_impl(tensor, f32_conv_buf, workers, nelements, nthread);845 f32_data = (float *) f32_conv_buf.data();846 }847 848 LLAMA_LOG_INFO("converting to %s .. ", ggml_type_name(new_type));849 fflush(stdout);850 851 if (work.size() < (size_t)nelements * 4) {852 work.resize(nelements * 4); // upper bound on size853 }854 new_data = work.data();855 856 const int64_t n_per_row = tensor->ne[0];857 const int64_t nrows = tensor->ne[1];858 859 static const int64_t min_chunk_size = 32 * 512;860 const int64_t chunk_size = (n_per_row >= min_chunk_size ? n_per_row : n_per_row * ((min_chunk_size + n_per_row - 1)/n_per_row));861 862 const int64_t nelements_matrix = tensor->ne[0] * tensor->ne[1];863 const int64_t nchunk = (nelements_matrix + chunk_size - 1)/chunk_size;864 const int64_t nthread_use = nthread > 1 ? std::max((int64_t)1, std::min((int64_t)nthread, nchunk)) : 1;865 866 // quantize each expert separately since they have different importance matrices867 new_size = 0;868 for (int64_t i03 = 0; i03 < tensor->ne[2]; ++i03) {869 const float * f32_data_03 = f32_data + i03 * nelements_matrix;870 void * new_data_03 = (char *)new_data + ggml_row_size(new_type, n_per_row) * i03 * nrows;871 const float * imatrix_03 = imatrix ? imatrix + i03 * n_per_row : nullptr;872 873 new_size += llama_tensor_quantize_impl(new_type, f32_data_03, new_data_03, chunk_size, nrows, n_per_row, imatrix_03, workers, nthread_use);874 }875 LLAMA_LOG_INFO("size = %8.2f MiB -> %8.2f MiB\n", ggml_nbytes(tensor)/1024.0/1024.0, new_size/1024.0/1024.0);876 }877 total_size_org += ggml_nbytes(tensor);878 total_size_new += new_size;879 880 // update the gguf meta data as we go881 gguf_set_tensor_type(ctx_outs[cur_split].get(), name.c_str(), new_type);882 GGML_ASSERT(gguf_get_tensor_size(ctx_outs[cur_split].get(), gguf_find_tensor(ctx_outs[cur_split].get(), name.c_str())) == new_size);883 gguf_set_tensor_data(ctx_outs[cur_split].get(), name.c_str(), new_data);884 885 // write tensor data + padding886 fout.write((const char *) new_data, new_size);887 zeros(fout, GGML_PAD(new_size, align) - new_size);888 }889 close_ofstream();890 891 LLAMA_LOG_INFO("%s: model size = %8.2f MB\n", __func__, total_size_org/1024.0/1024.0);892 LLAMA_LOG_INFO("%s: quant size = %8.2f MB\n", __func__, total_size_new/1024.0/1024.0);893 894 if (qs.n_fallback > 0) {895 LLAMA_LOG_WARN("%s: WARNING: %d of %d tensor(s) required fallback quantization\n",896 __func__, qs.n_fallback, qs.n_k_quantized + qs.n_fallback);897 }898}899 900//901// interface implementation902//903 904struct llama_model_quantize_params llama_model_quantize_default_params() {905 struct llama_model_quantize_params result = {906 /*.nthread =*/ 0,907 /*.ftype =*/ LLAMA_FTYPE_MOSTLY_Q5_1,908 /*.output_tensor_type =*/ GGML_TYPE_COUNT,909 /*.token_embedding_type =*/ GGML_TYPE_COUNT,910 /*.allow_requantize =*/ false,911 /*.quantize_output_tensor =*/ true,912 /*.only_copy =*/ false,913 /*.pure =*/ false,914 /*.keep_split =*/ false,915 /*.imatrix =*/ nullptr,916 /*.kv_overrides =*/ nullptr,917 };918 919 return result;920}921 922uint32_t llama_model_quantize(923 const char * fname_inp,924 const char * fname_out,925 const llama_model_quantize_params * params) {926 try {927 llama_model_quantize_impl(fname_inp, fname_out, params);928 } catch (const std::exception & err) {929 LLAMA_LOG_ERROR("%s: failed to quantize: %s\n", __func__, err.what());930 return 1;931 }932 933 return 0;934}935 