Felipe97/llama-cpp-compiled
01.2k
1#include "llama.h"2 3#include "build-info.h"4#include "common.h"5 6#include "ggml.h"7#include "gguf.h"8 9#include <algorithm>10#include <cinttypes>11#include <climits>12#include <clocale>13#include <cstdio>14#include <cstdlib>15#include <stdexcept>16#include <cstring>17#include <fstream>18#include <string>19#include <vector>20 21#if defined(_WIN32)22 #include <windows.h>23 #ifndef PATH_MAX24 #define PATH_MAX MAX_PATH25 #endif26 #include <io.h>27#endif28 29enum split_operation : uint8_t {30 OP_NONE,31 OP_SPLIT,32 OP_MERGE,33};34 35enum split_mode : uint8_t {36 MODE_NONE,37 MODE_TENSOR,38 MODE_SIZE,39};40 41struct split_params {42 split_operation operation = OP_NONE;43 split_mode mode = MODE_NONE;44 size_t n_bytes_split = 0;45 int n_split_tensors = 128;46 std::string input;47 std::string output;48 bool no_tensor_first_split = false;49 bool dry_run = false;50 bool delete_splits = false;51};52 53static void split_print_usage(const char * executable) {54 const split_params default_params;55 printf("\n");56 printf("usage: %s [options] GGUF_IN GGUF_OUT\n", executable);57 printf("\n");58 printf("Apply a GGUF operation on IN to OUT.");59 printf("\n");60 printf("options:\n");61 printf(" -h, --help show this help message and exit\n");62 printf(" --version show version and build info\n");63 printf(" --split split GGUF to multiple GGUF (enabled by default)\n");64 printf(" --merge merge multiple GGUF to a single GGUF\n");65 printf(" --split-max-tensors max tensors in each split (default: %d)\n", default_params.n_split_tensors);66 printf(" --split-max-size N(M|G) max size per split\n");67 printf(" --no-tensor-first-split do not add tensors to the first split (disabled by default)\n");68 printf(" --dry-run only print out a split plan and exit, without writing any new files\n");69 printf(" --delete-splits delete the split files during merge to free up disk space WARNING: this option is unsafe and will leave you in an unrecoverable state if something fails during the merge\n");70 printf("\n");71}72 73// return convert string, for example "128M" or "4G" to number of bytes74static size_t split_str_to_n_bytes(std::string str) {75 size_t n_bytes = 0;76 int n;77 if (str.back() == 'M') {78 sscanf(str.c_str(), "%d", &n);79 n_bytes = (size_t)n * 1000 * 1000; // megabytes80 } else if (str.back() == 'G') {81 sscanf(str.c_str(), "%d", &n);82 n_bytes = (size_t)n * 1000 * 1000 * 1000; // gigabytes83 } else {84 throw std::invalid_argument("error: supported units are M (megabytes) or G (gigabytes), but got: " + std::string(1, str.back()));85 }86 if (n <= 0) {87 throw std::invalid_argument("error: size must be a positive value");88 }89 return n_bytes;90}91 92static void split_params_parse_ex(int argc, const char ** argv, split_params & params) {93 std::string arg;94 const std::string arg_prefix = "--";95 bool invalid_param = false;96 97 int arg_idx = 1;98 for (; arg_idx < argc && strncmp(argv[arg_idx], "--", 2) == 0; arg_idx++) {99 arg = argv[arg_idx];100 if (arg.compare(0, arg_prefix.size(), arg_prefix) == 0) {101 std::replace(arg.begin(), arg.end(), '_', '-');102 }103 104 bool arg_found = false;105 if (arg == "-h" || arg == "--help") {106 split_print_usage(argv[0]);107 exit(0);108 } else if (arg == "--version") {109 fprintf(stderr, "version: %s (build %d, commit %s)\n", llama_version(), llama_build_number(), llama_commit());110 fprintf(stderr, "built with %s for %s\n", llama_compiler(), llama_build_target());111 exit(0);112 } else if (arg == "--dry-run") {113 arg_found = true;114 params.dry_run = true;115 } else if (arg == "--no-tensor-first-split") {116 arg_found = true;117 params.no_tensor_first_split = true;118 } else if (arg == "--merge") {119 arg_found = true;120 if (params.operation != OP_NONE && params.operation != OP_MERGE) {121 throw std::invalid_argument("error: either --split or --merge can be specified, but not both");122 }123 params.operation = OP_MERGE;124 } else if (arg == "--split") {125 arg_found = true;126 if (params.operation != OP_NONE && params.operation != OP_SPLIT) {127 throw std::invalid_argument("error: either --split or --merge can be specified, but not both");128 }129 params.operation = OP_SPLIT;130 } else if (arg == "--split-max-tensors") {131 if (++arg_idx >= argc) {132 invalid_param = true;133 break;134 }135 arg_found = true;136 if (params.mode != MODE_NONE && params.mode != MODE_TENSOR) {137 throw std::invalid_argument("error: either --split-max-tensors or --split-max-size can be specified, but not both");138 }139 params.mode = MODE_TENSOR;140 params.n_split_tensors = atoi(argv[arg_idx]);141 } else if (arg == "--split-max-size") {142 if (++arg_idx >= argc) {143 invalid_param = true;144 break;145 }146 arg_found = true;147 if (params.mode != MODE_NONE && params.mode != MODE_SIZE) {148 throw std::invalid_argument("error: either --split-max-tensors or --split-max-size can be specified, but not both");149 }150 params.mode = MODE_SIZE;151 params.n_bytes_split = split_str_to_n_bytes(argv[arg_idx]);152 } else if (arg == "--delete-splits") {153 arg_found = true;154 params.delete_splits = true;155 }156 157 if (!arg_found) {158 throw std::invalid_argument("error: unknown argument: " + arg);159 }160 }161 162 // the operation is split if not specified163 if (params.operation == OP_NONE) {164 params.operation = OP_SPLIT;165 }166 // the split mode is by tensor if not specified167 if (params.mode == MODE_NONE) {168 params.mode = MODE_TENSOR;169 }170 171 if (invalid_param) {172 throw std::invalid_argument("error: invalid parameter for argument: " + arg);173 }174 175 if (argc - arg_idx != 2) {176 throw std::invalid_argument("error: bad arguments");177 }178 179 params.input = argv[arg_idx++];180 params.output = argv[arg_idx++];181}182 183static bool split_params_parse(int argc, const char ** argv, split_params & params) {184 bool result = true;185 try {186 split_params_parse_ex(argc, argv, params);187 }188 catch (const std::invalid_argument & ex) {189 fprintf(stderr, "%s\n", ex.what());190 split_print_usage(argv[0]);191 exit(EXIT_FAILURE);192 }193 return result;194}195 196static void zeros(std::ofstream & file, size_t n) {197 char zero = 0;198 for (size_t i = 0; i < n; ++i) {199 file.write(&zero, 1);200 }201}202 203struct split_strategy {204 const split_params params;205 std::ifstream & f_input;206 struct gguf_context * ctx_gguf;207 struct ggml_context * ctx_meta = NULL;208 const int n_tensors;209 210 // one ctx_out per one output file211 std::vector<struct gguf_context *> ctx_outs;212 213 // temporary buffer for reading in tensor data214 std::vector<uint8_t> read_buf;215 216 split_strategy(const split_params & params,217 std::ifstream & f_input,218 struct gguf_context * ctx_gguf,219 struct ggml_context * ctx_meta) :220 params(params),221 f_input(f_input),222 ctx_gguf(ctx_gguf),223 ctx_meta(ctx_meta),224 n_tensors(gguf_get_n_tensors(ctx_gguf)) {225 226 // because we need to know list of tensors for each file in advance, we will build all the ctx_out for all output splits227 int i_split = -1;228 struct gguf_context * ctx_out = NULL;229 auto new_ctx_out = [&](bool allow_no_tensors) {230 i_split++;231 if (ctx_out != NULL) {232 if (gguf_get_n_tensors(ctx_out) == 0 && !allow_no_tensors) {233 fprintf(stderr, "error: one of splits have 0 tensors. Maybe size or tensors limit is too small\n");234 exit(EXIT_FAILURE);235 }236 ctx_outs.push_back(ctx_out);237 }238 ctx_out = gguf_init_empty();239 // Save all metadata in first split only240 if (i_split == 0) {241 gguf_set_kv(ctx_out, ctx_gguf);242 }243 gguf_set_val_u16(ctx_out, LLM_KV_SPLIT_NO, i_split);244 gguf_set_val_u16(ctx_out, LLM_KV_SPLIT_COUNT, 0); // placeholder245 gguf_set_val_i32(ctx_out, LLM_KV_SPLIT_TENSORS_COUNT, n_tensors);246 };247 248 // initialize ctx_out for the first split249 new_ctx_out(false);250 251 // skip first split if no_tensor_first_split is set252 if (params.no_tensor_first_split) {253 new_ctx_out(true);254 }255 256 // process tensors one by one257 size_t curr_tensors_size = 0; // current size by counting only tensors size (without metadata)258 for (int i = 0; i < n_tensors; ++i) {259 struct ggml_tensor * t = ggml_get_tensor(ctx_meta, gguf_get_tensor_name(ctx_gguf, i));260 // calculate the "imaginary" size = the current size + next tensor size261 size_t n_bytes = GGML_PAD(ggml_nbytes(t), GGUF_DEFAULT_ALIGNMENT);262 size_t next_tensors_size = curr_tensors_size + n_bytes;263 if (should_split(i, next_tensors_size)) {264 new_ctx_out(false);265 curr_tensors_size = n_bytes;266 } else {267 curr_tensors_size = next_tensors_size;268 }269 gguf_add_tensor(ctx_out, t);270 }271 272 // push the last ctx_out273 ctx_outs.push_back(ctx_out);274 275 // set the correct n_split for all ctx_out276 for (auto & ctx : ctx_outs) {277 gguf_set_val_u16(ctx, LLM_KV_SPLIT_COUNT, ctx_outs.size());278 }279 }280 281 ~split_strategy() {282 for (auto & ctx_out : ctx_outs) {283 gguf_free(ctx_out);284 }285 }286 287 bool should_split(int i_tensor, size_t next_size) {288 if (params.mode == MODE_SIZE) {289 // split by max size per file290 return next_size > params.n_bytes_split;291 } else if (params.mode == MODE_TENSOR) {292 // split by number of tensors per file293 return i_tensor > 0 && i_tensor < n_tensors && i_tensor % params.n_split_tensors == 0;294 }295 // should never happen296 GGML_ABORT("invalid mode");297 }298 299 void print_info() {300 printf("n_split: %zu\n", ctx_outs.size());301 int i_split = 0;302 for (auto & ctx_out : ctx_outs) {303 // re-calculate the real gguf size for each split (= metadata size + total size of all tensors)304 size_t total_size = gguf_get_meta_size(ctx_out);305 for (int i = 0; i < gguf_get_n_tensors(ctx_out); ++i) {306 struct ggml_tensor * t = ggml_get_tensor(ctx_meta, gguf_get_tensor_name(ctx_out, i));307 total_size += ggml_nbytes(t);308 }309 total_size = total_size / 1000 / 1000; // convert to megabytes310 printf("split %05d: n_tensors = %" PRIi64 ", total_size = %zuM\n", i_split + 1, gguf_get_n_tensors(ctx_out), total_size);311 i_split++;312 }313 }314 315 void write() {316 int i_split = 0;317 int n_split = ctx_outs.size();318 for (auto & ctx_out : ctx_outs) {319 // construct file path320 char split_path[PATH_MAX] = {0};321 llama_split_path(split_path, sizeof(split_path), params.output.c_str(), i_split, n_split);322 323 // open the output file324 printf("Writing file %s ... ", split_path);325 fflush(stdout);326 std::ofstream fout = std::ofstream(split_path, std::ios::binary);327 fout.exceptions(std::ofstream::failbit); // fail fast on write errors328 329 // write metadata330 std::vector<uint8_t> data(gguf_get_meta_size(ctx_out));331 gguf_get_meta_data(ctx_out, data.data());332 fout.write((const char *)data.data(), data.size());333 334 // write tensors335 for (int i = 0; i < gguf_get_n_tensors(ctx_out); ++i) {336 // read tensor meta and prepare buffer337 const char * t_name = gguf_get_tensor_name(ctx_out, i);338 struct ggml_tensor * t = ggml_get_tensor(ctx_meta, t_name);339 auto n_bytes = ggml_nbytes(t);340 read_buf.resize(n_bytes);341 342 // calculate offset343 auto i_tensor_in = gguf_find_tensor(ctx_gguf, t_name); // idx of tensor in the input file344 auto offset = gguf_get_data_offset(ctx_gguf) + gguf_get_tensor_offset(ctx_gguf, i_tensor_in);345 346 // copy tensor from input to output file347 copy_file_to_file(f_input, fout, offset, n_bytes);348 zeros(fout, GGML_PAD(n_bytes, GGUF_DEFAULT_ALIGNMENT) - n_bytes);349 }350 351 printf("done\n");352 // close the file353 fout.close();354 i_split++;355 }356 }357 358 void copy_file_to_file(std::ifstream & f_in, std::ofstream & f_out, const size_t in_offset, const size_t len) {359 // TODO: detect OS and use copy_file_range() here for better performance360 if (read_buf.size() < len) {361 read_buf.resize(len);362 }363 f_in.seekg(in_offset);364 f_in.read((char *)read_buf.data(), len);365 f_out.write((const char *)read_buf.data(), len);366 }367};368 369static void gguf_split(const split_params & split_params) {370 struct ggml_context * ctx_meta = NULL;371 372 struct gguf_init_params params = {373 /*.no_alloc = */ true,374 /*.ctx = */ &ctx_meta,375 };376 377 std::ifstream f_input(split_params.input.c_str(), std::ios::binary);378 if (!f_input.is_open()) {379 fprintf(stderr, "%s: failed to open input GGUF from %s\n", __func__, split_params.input.c_str());380 exit(EXIT_FAILURE);381 }382 383 auto * ctx_gguf = gguf_init_from_file(split_params.input.c_str(), params);384 if (!ctx_gguf) {385 fprintf(stderr, "%s: failed to load input GGUF from %s\n", __func__, split_params.input.c_str());386 exit(EXIT_FAILURE);387 }388 389 // prepare the strategy390 split_strategy strategy(split_params, f_input, ctx_gguf, ctx_meta);391 int n_split = strategy.ctx_outs.size();392 strategy.print_info();393 394 if (!split_params.dry_run) {395 // write all output splits396 strategy.write();397 }398 399 // done, clean up400 gguf_free(ctx_gguf);401 f_input.close();402 403 fprintf(stderr, "%s: %d gguf split written with a total of %d tensors.\n",404 __func__, n_split, strategy.n_tensors);405}406 407static void gguf_merge(const split_params & split_params) {408 fprintf(stderr, "%s: %s -> %s\n",409 __func__, split_params.input.c_str(),410 split_params.output.c_str());411 int n_split = 1;412 int total_tensors = 0;413 414 // avoid overwriting existing output file415 if (std::ifstream(split_params.output.c_str())) {416 fprintf(stderr, "%s: output file %s already exists\n", __func__, split_params.output.c_str());417 exit(EXIT_FAILURE);418 }419 420 421 auto * ctx_out = gguf_init_empty();422 423 std::vector<uint8_t> read_data;424 std::vector<ggml_context *> ctx_metas;425 std::vector<gguf_context *> ctx_ggufs;426 427 char split_path[PATH_MAX] = {0};428 strncpy(split_path, split_params.input.c_str(), sizeof(split_path) - 1);429 char split_prefix[PATH_MAX] = {0};430 431 // First pass to find KV and tensors metadata432 for (int i_split = 0; i_split < n_split; i_split++) {433 struct ggml_context * ctx_meta = NULL;434 435 struct gguf_init_params params = {436 /*.no_alloc = */ true,437 /*.ctx = */ &ctx_meta,438 };439 440 if (i_split > 0) {441 llama_split_path(split_path, sizeof(split_path), split_prefix, i_split, n_split);442 }443 fprintf(stderr, "%s: reading metadata %s ...", __func__, split_path);444 445 auto * ctx_gguf = gguf_init_from_file(split_path, params);446 if (!ctx_gguf) {447 fprintf(stderr, "\n%s: failed to load input GGUF from %s\n", __func__, split_params.input.c_str());448 exit(EXIT_FAILURE);449 }450 ctx_ggufs.push_back(ctx_gguf);451 ctx_metas.push_back(ctx_meta);452 453 if (i_split == 0) {454 auto key_n_split = gguf_find_key(ctx_gguf, LLM_KV_SPLIT_COUNT);455 if (key_n_split < 0) {456 fprintf(stderr,457 "\n%s: input file does not contain %s metadata\n",458 __func__,459 LLM_KV_SPLIT_COUNT);460 gguf_free(ctx_gguf);461 ggml_free(ctx_meta);462 gguf_free(ctx_out);463 exit(EXIT_FAILURE);464 }465 466 n_split = gguf_get_val_u16(ctx_gguf, key_n_split);467 if (n_split < 1) {468 fprintf(stderr,469 "\n%s: input file does not contain a valid split count %d\n",470 __func__,471 n_split);472 gguf_free(ctx_gguf);473 ggml_free(ctx_meta);474 gguf_free(ctx_out);475 exit(EXIT_FAILURE);476 }477 478 // Verify the file naming and extract split_prefix479 if (!llama_split_prefix(split_prefix, sizeof (split_prefix), split_path, i_split, n_split)) {480 fprintf(stderr, "\n%s: unexpected input file name: %s"481 " i_split=%d"482 " n_split=%d\n", __func__,483 split_path, i_split, n_split);484 gguf_free(ctx_gguf);485 ggml_free(ctx_meta);486 gguf_free(ctx_out);487 exit(EXIT_FAILURE);488 }489 490 // Do not trigger merge if we try to merge again the output491 gguf_set_val_u16(ctx_gguf, LLM_KV_SPLIT_COUNT, 0);492 493 // Set metadata from the first split494 gguf_set_kv(ctx_out, ctx_gguf);495 }496 497 auto n_tensors = gguf_get_n_tensors(ctx_gguf);498 for (int i_tensor = 0; i_tensor < n_tensors; i_tensor++) {499 const char * t_name = gguf_get_tensor_name(ctx_gguf, i_tensor);500 struct ggml_tensor * t = ggml_get_tensor(ctx_meta, t_name);501 gguf_add_tensor(ctx_out, t);502 }503 total_tensors += n_tensors;504 505 fprintf(stderr, "\033[3Ddone\n");506 }507 std::ofstream fout;508 if (!split_params.dry_run) {509 fout.open(split_params.output.c_str(), std::ios::binary);510 fout.exceptions(std::ofstream::failbit); // fail fast on write errors511 // placeholder for the meta data512 auto meta_size = gguf_get_meta_size(ctx_out);513 ::zeros(fout, meta_size);514 }515 516 // Write tensors data517 bool merge_error = false;518 for (int i_split = 0; i_split < n_split; i_split++) {519 llama_split_path(split_path, sizeof(split_path), split_prefix, i_split, n_split);520 std::ifstream f_input(split_path, std::ios::binary);521 if (!f_input.is_open()) {522 fprintf(stderr, "%s: failed to open input GGUF from %s\n", __func__, split_path);523 for (uint32_t i = 0; i < ctx_ggufs.size(); i++) {524 gguf_free(ctx_ggufs[i]);525 ggml_free(ctx_metas[i]);526 }527 gguf_free(ctx_out);528 if (!split_params.dry_run) {529 fout.close();530 }531 exit(EXIT_FAILURE);532 }533 fprintf(stderr, "%s: writing tensors %s ...", __func__, split_path);534 535 auto * ctx_gguf = ctx_ggufs[i_split];536 auto * ctx_meta = ctx_metas[i_split];537 538 auto n_tensors = gguf_get_n_tensors(ctx_gguf);539 for (int i_tensor = 0; i_tensor < n_tensors; i_tensor++) {540 const char * t_name = gguf_get_tensor_name(ctx_gguf, i_tensor);541 struct ggml_tensor * t = ggml_get_tensor(ctx_meta, t_name);542 543 auto n_bytes = ggml_nbytes(t);544 545 if (read_data.size() < n_bytes) {546 read_data.resize(n_bytes);547 }548 549 auto offset = gguf_get_data_offset(ctx_gguf) + gguf_get_tensor_offset(ctx_gguf, i_tensor);550 f_input.seekg(offset);551 f_input.read((char *)read_data.data(), n_bytes);552 if (!split_params.dry_run) {553 // write tensor data + padding554 fout.write((const char *)read_data.data(), n_bytes);555 zeros(fout, GGML_PAD(n_bytes, GGUF_DEFAULT_ALIGNMENT) - n_bytes);556 }557 }558 559 gguf_free(ctx_gguf);560 ggml_free(ctx_meta);561 f_input.close();562 fprintf(stderr, "\033[3Ddone\n");563 564 if (!split_params.dry_run && split_params.delete_splits) {565 int delete_result = std::remove(split_path);566 if (delete_result != 0) {567 merge_error = true;568 fprintf(stderr, "error: failed to delete %s\n", split_path);569 } else {570 fprintf(stderr, "%s: deleted file %s\n", __func__, split_path);571 }572 }573 }574 575 if (!split_params.dry_run) {576 // go back to beginning of file and write the updated metadata577 fout.seekp(0);578 std::vector<uint8_t> data(gguf_get_meta_size(ctx_out));579 gguf_get_meta_data(ctx_out, data.data());580 fout.write((const char *)data.data(), data.size());581 fout.close();582 }583 gguf_free(ctx_out);584 585 fprintf(stderr, "%s: %s merged from %d split with %d tensors.\n",586 __func__, split_params.output.c_str(), n_split, total_tensors);587 588 if (merge_error) {589 exit(EXIT_FAILURE);590 }591}592 593int main(int argc, const char ** argv) {594 std::setlocale(LC_NUMERIC, "C");595 596 split_params params;597 split_params_parse(argc, argv, params);598 599 switch (params.operation) {600 case OP_SPLIT: gguf_split(params);601 break;602 case OP_MERGE: gguf_merge(params);603 break;604 default: split_print_usage(argv[0]);605 exit(EXIT_FAILURE);606 }607 608 return 0;609}610 