KBaba7/llama.cpp
0
1#include "common.h"2#include "llama.h"3 4#include <cstdio>5#include <cstring>6#include <vector>7#include <string>8#include <unordered_map>9#include <fstream>10#include <cmath>11 12struct quant_option {13 std::string name;14 llama_ftype ftype;15 std::string desc;16};17 18static const std::vector<struct quant_option> QUANT_OPTIONS = {19 { "Q4_0", LLAMA_FTYPE_MOSTLY_Q4_0, " 4.34G, +0.4685 ppl @ Llama-3-8B", },20 { "Q4_1", LLAMA_FTYPE_MOSTLY_Q4_1, " 4.78G, +0.4511 ppl @ Llama-3-8B", },21 { "Q5_0", LLAMA_FTYPE_MOSTLY_Q5_0, " 5.21G, +0.1316 ppl @ Llama-3-8B", },22 { "Q5_1", LLAMA_FTYPE_MOSTLY_Q5_1, " 5.65G, +0.1062 ppl @ Llama-3-8B", },23 { "IQ2_XXS", LLAMA_FTYPE_MOSTLY_IQ2_XXS, " 2.06 bpw quantization", },24 { "IQ2_XS", LLAMA_FTYPE_MOSTLY_IQ2_XS, " 2.31 bpw quantization", },25 { "IQ2_S", LLAMA_FTYPE_MOSTLY_IQ2_S, " 2.5 bpw quantization", },26 { "IQ2_M", LLAMA_FTYPE_MOSTLY_IQ2_M, " 2.7 bpw quantization", },27 { "IQ1_S", LLAMA_FTYPE_MOSTLY_IQ1_S, " 1.56 bpw quantization", },28 { "IQ1_M", LLAMA_FTYPE_MOSTLY_IQ1_M, " 1.75 bpw quantization", },29 { "TQ1_0", LLAMA_FTYPE_MOSTLY_TQ1_0, " 1.69 bpw ternarization", },30 { "TQ2_0", LLAMA_FTYPE_MOSTLY_TQ2_0, " 2.06 bpw ternarization", },31 { "Q2_K", LLAMA_FTYPE_MOSTLY_Q2_K, " 2.96G, +3.5199 ppl @ Llama-3-8B", },32 { "Q2_K_S", LLAMA_FTYPE_MOSTLY_Q2_K_S, " 2.96G, +3.1836 ppl @ Llama-3-8B", },33 { "IQ3_XXS", LLAMA_FTYPE_MOSTLY_IQ3_XXS, " 3.06 bpw quantization", },34 { "IQ3_S", LLAMA_FTYPE_MOSTLY_IQ3_S, " 3.44 bpw quantization", },35 { "IQ3_M", LLAMA_FTYPE_MOSTLY_IQ3_M, " 3.66 bpw quantization mix", },36 { "Q3_K", LLAMA_FTYPE_MOSTLY_Q3_K_M, "alias for Q3_K_M" },37 { "IQ3_XS", LLAMA_FTYPE_MOSTLY_IQ3_XS, " 3.3 bpw quantization", },38 { "Q3_K_S", LLAMA_FTYPE_MOSTLY_Q3_K_S, " 3.41G, +1.6321 ppl @ Llama-3-8B", },39 { "Q3_K_M", LLAMA_FTYPE_MOSTLY_Q3_K_M, " 3.74G, +0.6569 ppl @ Llama-3-8B", },40 { "Q3_K_L", LLAMA_FTYPE_MOSTLY_Q3_K_L, " 4.03G, +0.5562 ppl @ Llama-3-8B", },41 { "IQ4_NL", LLAMA_FTYPE_MOSTLY_IQ4_NL, " 4.50 bpw non-linear quantization", },42 { "IQ4_XS", LLAMA_FTYPE_MOSTLY_IQ4_XS, " 4.25 bpw non-linear quantization", },43 { "Q4_K", LLAMA_FTYPE_MOSTLY_Q4_K_M, "alias for Q4_K_M", },44 { "Q4_K_S", LLAMA_FTYPE_MOSTLY_Q4_K_S, " 4.37G, +0.2689 ppl @ Llama-3-8B", },45 { "Q4_K_M", LLAMA_FTYPE_MOSTLY_Q4_K_M, " 4.58G, +0.1754 ppl @ Llama-3-8B", },46 { "Q5_K", LLAMA_FTYPE_MOSTLY_Q5_K_M, "alias for Q5_K_M", },47 { "Q5_K_S", LLAMA_FTYPE_MOSTLY_Q5_K_S, " 5.21G, +0.1049 ppl @ Llama-3-8B", },48 { "Q5_K_M", LLAMA_FTYPE_MOSTLY_Q5_K_M, " 5.33G, +0.0569 ppl @ Llama-3-8B", },49 { "Q6_K", LLAMA_FTYPE_MOSTLY_Q6_K, " 6.14G, +0.0217 ppl @ Llama-3-8B", },50 { "Q8_0", LLAMA_FTYPE_MOSTLY_Q8_0, " 7.96G, +0.0026 ppl @ Llama-3-8B", },51 { "F16", LLAMA_FTYPE_MOSTLY_F16, "14.00G, +0.0020 ppl @ Mistral-7B", },52 { "BF16", LLAMA_FTYPE_MOSTLY_BF16, "14.00G, -0.0050 ppl @ Mistral-7B", },53 { "F32", LLAMA_FTYPE_ALL_F32, "26.00G @ 7B", },54 // Note: Ensure COPY comes after F32 to avoid ftype 0 from matching.55 { "COPY", LLAMA_FTYPE_ALL_F32, "only copy tensors, no quantizing", },56};57 58static const char * const LLM_KV_QUANTIZE_IMATRIX_FILE = "quantize.imatrix.file";59static const char * const LLM_KV_QUANTIZE_IMATRIX_DATASET = "quantize.imatrix.dataset";60static const char * const LLM_KV_QUANTIZE_IMATRIX_N_ENTRIES = "quantize.imatrix.entries_count";61static const char * const LLM_KV_QUANTIZE_IMATRIX_N_CHUNKS = "quantize.imatrix.chunks_count";62 63static bool striequals(const char * a, const char * b) {64 while (*a && *b) {65 if (std::tolower(*a) != std::tolower(*b)) {66 return false;67 }68 a++; b++;69 }70 return *a == *b;71}72 73static bool try_parse_ftype(const std::string & ftype_str_in, llama_ftype & ftype, std::string & ftype_str_out) {74 std::string ftype_str;75 76 for (auto ch : ftype_str_in) {77 ftype_str.push_back(std::toupper(ch));78 }79 for (auto & it : QUANT_OPTIONS) {80 if (striequals(it.name.c_str(), ftype_str.c_str())) {81 ftype = it.ftype;82 ftype_str_out = it.name;83 return true;84 }85 }86 try {87 int ftype_int = std::stoi(ftype_str);88 for (auto & it : QUANT_OPTIONS) {89 if (it.ftype == ftype_int) {90 ftype = it.ftype;91 ftype_str_out = it.name;92 return true;93 }94 }95 }96 catch (...) {97 // stoi failed98 }99 return false;100}101 102// usage:103// ./llama-quantize [--allow-requantize] [--leave-output-tensor] [--pure] models/llama/ggml-model.gguf [models/llama/ggml-model-quant.gguf] type [nthreads]104//105[[noreturn]]106static void usage(const char * executable) {107 printf("usage: %s [--help] [--allow-requantize] [--leave-output-tensor] [--pure] [--imatrix] [--include-weights] [--exclude-weights] [--output-tensor-type] [--token-embedding-type] [--override-kv] model-f32.gguf [model-quant.gguf] type [nthreads]\n\n", executable);108 printf(" --allow-requantize: Allows requantizing tensors that have already been quantized. Warning: This can severely reduce quality compared to quantizing from 16bit or 32bit\n");109 printf(" --leave-output-tensor: Will leave output.weight un(re)quantized. Increases model size but may also increase quality, especially when requantizing\n");110 printf(" --pure: Disable k-quant mixtures and quantize all tensors to the same type\n");111 printf(" --imatrix file_name: use data in file_name as importance matrix for quant optimizations\n");112 printf(" --include-weights tensor_name: use importance matrix for this/these tensor(s)\n");113 printf(" --exclude-weights tensor_name: use importance matrix for this/these tensor(s)\n");114 printf(" --output-tensor-type ggml_type: use this ggml_type for the output.weight tensor\n");115 printf(" --token-embedding-type ggml_type: use this ggml_type for the token embeddings tensor\n");116 printf(" --keep-split: will generate quantized model in the same shards as input\n");117 printf(" --override-kv KEY=TYPE:VALUE\n");118 printf(" Advanced option to override model metadata by key in the quantized model. May be specified multiple times.\n");119 printf("Note: --include-weights and --exclude-weights cannot be used together\n");120 printf("\nAllowed quantization types:\n");121 for (auto & it : QUANT_OPTIONS) {122 if (it.name != "COPY") {123 printf(" %2d or ", it.ftype);124 } else {125 printf(" ");126 }127 printf("%-7s : %s\n", it.name.c_str(), it.desc.c_str());128 }129 exit(1);130}131 132static int load_imatrix(const std::string & imatrix_file, std::string & imatrix_dataset, std::unordered_map<std::string, std::vector<float>> & imatrix_data) {133 std::ifstream in(imatrix_file.c_str(), std::ios::binary);134 if (!in) {135 printf("%s: failed to open %s\n",__func__, imatrix_file.c_str());136 exit(1);137 }138 int n_entries;139 in.read((char *)&n_entries, sizeof(n_entries));140 if (in.fail() || n_entries < 1) {141 printf("%s: no data in file %s\n", __func__, imatrix_file.c_str());142 exit(1);143 }144 for (int i = 0; i < n_entries; ++i) {145 int len; in.read((char *)&len, sizeof(len));146 std::vector<char> name_as_vec(len+1);147 in.read((char *)name_as_vec.data(), len);148 if (in.fail()) {149 printf("%s: failed reading name for entry %d from %s\n", __func__, i+1, imatrix_file.c_str());150 exit(1);151 }152 name_as_vec[len] = 0;153 std::string name{name_as_vec.data()};154 auto & e = imatrix_data[name];155 int ncall;156 in.read((char *)&ncall, sizeof(ncall));157 int nval;158 in.read((char *)&nval, sizeof(nval));159 if (in.fail() || nval < 1) {160 printf("%s: failed reading number of values for entry %d\n", __func__, i);161 imatrix_data = {};162 exit(1);163 }164 e.resize(nval);165 in.read((char *)e.data(), nval*sizeof(float));166 if (in.fail()) {167 printf("%s: failed reading data for entry %d\n", __func__, i);168 imatrix_data = {};169 exit(1);170 }171 if (ncall > 0) {172 for (auto& v : e) v /= ncall;173 }174 175 if (getenv("LLAMA_TRACE")) {176 printf("%s: loaded data (size = %6d, ncall = %6d) for '%s'\n", __func__, int(e.size()), ncall, name.c_str());177 }178 }179 180 // latest imatrix version contains the dataset filename at the end of the file181 int m_last_call = 0;182 if (in.peek() != EOF) {183 in.read((char *)&m_last_call, sizeof(m_last_call));184 int dataset_len;185 in.read((char *)&dataset_len, sizeof(dataset_len));186 std::vector<char> dataset_as_vec(dataset_len);187 in.read(dataset_as_vec.data(), dataset_len);188 imatrix_dataset.assign(dataset_as_vec.begin(), dataset_as_vec.end());189 printf("%s: imatrix dataset='%s'\n", __func__, imatrix_dataset.c_str());190 }191 printf("%s: loaded %d importance matrix entries from %s computed on %d chunks\n", __func__, int(imatrix_data.size()), imatrix_file.c_str(), m_last_call);192 return m_last_call;193}194 195static int prepare_imatrix(const std::string & imatrix_file,196 std::string & imatrix_dataset,197 const std::vector<std::string> & included_weights,198 const std::vector<std::string> & excluded_weights,199 std::unordered_map<std::string, std::vector<float>> & imatrix_data) {200 int m_last_call = -1;201 if (!imatrix_file.empty()) {202 m_last_call = load_imatrix(imatrix_file, imatrix_dataset, imatrix_data);203 }204 if (imatrix_data.empty()) {205 return m_last_call;206 }207 if (!excluded_weights.empty()) {208 for (auto& name : excluded_weights) {209 for (auto it = imatrix_data.begin(); it != imatrix_data.end(); ) {210 auto pos = it->first.find(name);211 if (pos != std::string::npos) it = imatrix_data.erase(it);212 else ++it;213 }214 }215 }216 if (!included_weights.empty()) {217 std::unordered_map<std::string, std::vector<float>> tmp;218 for (auto& name : included_weights) {219 for (auto& e : imatrix_data) {220 auto pos = e.first.find(name);221 if (pos != std::string::npos) {222 tmp.emplace(std::move(e));223 }224 }225 }226 imatrix_data = std::move(tmp);227 }228 if (!imatrix_data.empty()) {229 printf("%s: have %d importance matrix entries\n", __func__, int(imatrix_data.size()));230 }231 return m_last_call;232}233 234static ggml_type parse_ggml_type(const char * arg) {235 for (int i = 0; i < GGML_TYPE_COUNT; ++i) {236 auto type = (ggml_type)i;237 const auto * name = ggml_type_name(type);238 if (name && striequals(name, arg)) {239 return type;240 }241 }242 fprintf(stderr, "%s: invalid ggml_type '%s'\n", __func__, arg);243 return GGML_TYPE_COUNT;244}245 246int main(int argc, char ** argv) {247 if (argc < 3) {248 usage(argv[0]);249 }250 251 llama_model_quantize_params params = llama_model_quantize_default_params();252 253 int arg_idx = 1;254 std::string imatrix_file;255 std::vector<std::string> included_weights, excluded_weights;256 std::vector<llama_model_kv_override> kv_overrides;257 258 for (; arg_idx < argc && strncmp(argv[arg_idx], "--", 2) == 0; arg_idx++) {259 if (strcmp(argv[arg_idx], "--leave-output-tensor") == 0) {260 params.quantize_output_tensor = false;261 } else if (strcmp(argv[arg_idx], "--output-tensor-type") == 0) {262 if (arg_idx < argc-1) {263 params.output_tensor_type = parse_ggml_type(argv[++arg_idx]);264 if (params.output_tensor_type == GGML_TYPE_COUNT) {265 usage(argv[0]);266 }267 } else {268 usage(argv[0]);269 }270 } else if (strcmp(argv[arg_idx], "--token-embedding-type") == 0) {271 if (arg_idx < argc-1) {272 params.token_embedding_type = parse_ggml_type(argv[++arg_idx]);273 if (params.token_embedding_type == GGML_TYPE_COUNT) {274 usage(argv[0]);275 }276 } else {277 usage(argv[0]);278 }279 } else if (strcmp(argv[arg_idx], "--override-kv") == 0) {280 if (arg_idx == argc-1 || !string_parse_kv_override(argv[++arg_idx], kv_overrides)) {281 usage(argv[0]);282 }283 } else if (strcmp(argv[arg_idx], "--allow-requantize") == 0) {284 params.allow_requantize = true;285 } else if (strcmp(argv[arg_idx], "--pure") == 0) {286 params.pure = true;287 } else if (strcmp(argv[arg_idx], "--imatrix") == 0) {288 if (arg_idx < argc-1) {289 imatrix_file = argv[++arg_idx];290 } else {291 usage(argv[0]);292 }293 } else if (strcmp(argv[arg_idx], "--include-weights") == 0) {294 if (arg_idx < argc-1) {295 included_weights.emplace_back(argv[++arg_idx]);296 } else {297 usage(argv[0]);298 }299 } else if (strcmp(argv[arg_idx], "--exclude-weights") == 0) {300 if (arg_idx < argc-1) {301 excluded_weights.emplace_back(argv[++arg_idx]);302 } else {303 usage(argv[0]);304 }305 } else if (strcmp(argv[arg_idx], "--keep-split") == 0) {306 params.keep_split = true;307 } else {308 usage(argv[0]);309 }310 }311 312 if (argc - arg_idx < 2) {313 printf("%s: bad arguments\n", argv[0]);314 usage(argv[0]);315 }316 if (!included_weights.empty() && !excluded_weights.empty()) {317 usage(argv[0]);318 }319 320 std::string imatrix_dataset;321 std::unordered_map<std::string, std::vector<float>> imatrix_data;322 int m_last_call = prepare_imatrix(imatrix_file, imatrix_dataset, included_weights, excluded_weights, imatrix_data);323 if (!imatrix_data.empty()) {324 params.imatrix = &imatrix_data;325 {326 llama_model_kv_override kvo;327 std::strcpy(kvo.key, LLM_KV_QUANTIZE_IMATRIX_FILE);328 kvo.tag = LLAMA_KV_OVERRIDE_TYPE_STR;329 strncpy(kvo.val_str, imatrix_file.c_str(), 127);330 kvo.val_str[127] = '\0';331 kv_overrides.emplace_back(std::move(kvo));332 }333 if (!imatrix_dataset.empty()) {334 llama_model_kv_override kvo;335 std::strcpy(kvo.key, LLM_KV_QUANTIZE_IMATRIX_DATASET);336 kvo.tag = LLAMA_KV_OVERRIDE_TYPE_STR;337 strncpy(kvo.val_str, imatrix_dataset.c_str(), 127);338 kvo.val_str[127] = '\0';339 kv_overrides.emplace_back(std::move(kvo));340 }341 342 {343 llama_model_kv_override kvo;344 std::strcpy(kvo.key, LLM_KV_QUANTIZE_IMATRIX_N_ENTRIES);345 kvo.tag = LLAMA_KV_OVERRIDE_TYPE_INT;346 kvo.val_i64 = imatrix_data.size();347 kv_overrides.emplace_back(std::move(kvo));348 }349 350 if (m_last_call > 0) {351 llama_model_kv_override kvo;352 std::strcpy(kvo.key, LLM_KV_QUANTIZE_IMATRIX_N_CHUNKS);353 kvo.tag = LLAMA_KV_OVERRIDE_TYPE_INT;354 kvo.val_i64 = m_last_call;355 kv_overrides.emplace_back(std::move(kvo));356 }357 }358 if (!kv_overrides.empty()) {359 kv_overrides.emplace_back();360 kv_overrides.back().key[0] = 0;361 params.kv_overrides = &kv_overrides;362 }363 364 llama_backend_init();365 366 // parse command line arguments367 const std::string fname_inp = argv[arg_idx];368 arg_idx++;369 std::string fname_out;370 371 std::string ftype_str;372 std::string suffix = ".gguf";373 if (try_parse_ftype(argv[arg_idx], params.ftype, ftype_str)) {374 std::string fpath;375 const size_t pos = fname_inp.find_last_of("/\\");376 if (pos != std::string::npos) {377 fpath = fname_inp.substr(0, pos + 1);378 }379 380 // export as [inp path]/ggml-model-[ftype]. Only add extension if there is no splitting381 fname_out = fpath + "ggml-model-" + ftype_str;382 if (!params.keep_split) {383 fname_out += suffix;384 }385 arg_idx++;386 if (ftype_str == "COPY") {387 params.only_copy = true;388 }389 } else {390 fname_out = argv[arg_idx];391 if (params.keep_split && fname_out.find(suffix) != std::string::npos) {392 fname_out = fname_out.substr(0, fname_out.length() - suffix.length());393 }394 arg_idx++;395 396 if (argc <= arg_idx) {397 fprintf(stderr, "%s: missing ftype\n", __func__);398 return 1;399 }400 if (!try_parse_ftype(argv[arg_idx], params.ftype, ftype_str)) {401 fprintf(stderr, "%s: invalid ftype '%s'\n", __func__, argv[3]);402 return 1;403 }404 if (ftype_str == "COPY") {405 params.only_copy = true;406 }407 arg_idx++;408 }409 410 // parse nthreads411 if (argc > arg_idx) {412 try {413 params.nthread = std::stoi(argv[arg_idx]);414 }415 catch (const std::exception & e) {416 fprintf(stderr, "%s: invalid nthread '%s' (%s)\n", __func__, argv[arg_idx], e.what());417 return 1;418 }419 }420 421 if ((params.ftype == LLAMA_FTYPE_MOSTLY_IQ2_XS || params.ftype == LLAMA_FTYPE_MOSTLY_IQ2_XXS ||422 params.ftype == LLAMA_FTYPE_MOSTLY_IQ2_S ||423 params.ftype == LLAMA_FTYPE_MOSTLY_Q2_K_S ||424 params.ftype == LLAMA_FTYPE_MOSTLY_IQ1_S ||425 params.ftype == LLAMA_FTYPE_MOSTLY_IQ1_M) && imatrix_data.empty()) {426 fprintf(stderr, "\n==========================================================================================================\n");427 fprintf(stderr, "Please do not use IQ1_S, IQ1_M, IQ2_S, IQ2_XXS, IQ2_XS or Q2_K_S quantization without an importance matrix\n");428 fprintf(stderr, "==========================================================================================================\n\n\n");429 return 1;430 }431 432 print_build_info();433 434 fprintf(stderr, "%s: quantizing '%s' to '%s' as %s", __func__, fname_inp.c_str(), fname_out.c_str(), ftype_str.c_str());435 if (params.nthread > 0) {436 fprintf(stderr, " using %d threads", params.nthread);437 }438 fprintf(stderr, "\n");439 440 const int64_t t_main_start_us = llama_time_us();441 442 int64_t t_quantize_us = 0;443 444 // load the model445 {446 const int64_t t_start_us = llama_time_us();447 448 if (llama_model_quantize(fname_inp.c_str(), fname_out.c_str(), ¶ms)) {449 fprintf(stderr, "%s: failed to quantize model from '%s'\n", __func__, fname_inp.c_str());450 return 1;451 }452 453 t_quantize_us = llama_time_us() - t_start_us;454 }455 456 // report timing457 {458 const int64_t t_main_end_us = llama_time_us();459 460 printf("\n");461 printf("%s: quantize time = %8.2f ms\n", __func__, t_quantize_us/1000.0);462 printf("%s: total time = %8.2f ms\n", __func__, (t_main_end_us - t_main_start_us)/1000.0);463 }464 465 llama_backend_free();466 467 return 0;468}469 