Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
gguf-split.cpp610 linesDownload Raw Back to gguf-split
1#include "llama.h"2 3#include "build-info.h"4#include "common.h"5 6#include "ggml.h"7#include "gguf.h"8 9#include <algorithm>10#include <cinttypes>11#include <climits>12#include <clocale>13#include <cstdio>14#include <cstdlib>15#include <stdexcept>16#include <cstring>17#include <fstream>18#include <string>19#include <vector>20 21#if defined(_WIN32)22    #include <windows.h>23    #ifndef PATH_MAX24        #define PATH_MAX MAX_PATH25    #endif26    #include <io.h>27#endif28 29enum split_operation : uint8_t {30    OP_NONE,31    OP_SPLIT,32    OP_MERGE,33};34 35enum split_mode : uint8_t {36    MODE_NONE,37    MODE_TENSOR,38    MODE_SIZE,39};40 41struct split_params {42    split_operation operation = OP_NONE;43    split_mode mode = MODE_NONE;44    size_t n_bytes_split = 0;45    int n_split_tensors = 128;46    std::string input;47    std::string output;48    bool no_tensor_first_split = false;49    bool dry_run = false;50    bool delete_splits = false;51};52 53static void split_print_usage(const char * executable) {54    const split_params default_params;55    printf("\n");56    printf("usage: %s [options] GGUF_IN GGUF_OUT\n", executable);57    printf("\n");58    printf("Apply a GGUF operation on IN to OUT.");59    printf("\n");60    printf("options:\n");61    printf("  -h, --help              show this help message and exit\n");62    printf("  --version               show version and build info\n");63    printf("  --split                 split GGUF to multiple GGUF (enabled by default)\n");64    printf("  --merge                 merge multiple GGUF to a single GGUF\n");65    printf("  --split-max-tensors     max tensors in each split (default: %d)\n", default_params.n_split_tensors);66    printf("  --split-max-size N(M|G) max size per split\n");67    printf("  --no-tensor-first-split do not add tensors to the first split (disabled by default)\n");68    printf("  --dry-run               only print out a split plan and exit, without writing any new files\n");69    printf("  --delete-splits         delete the split files during merge to free up disk space WARNING: this option is unsafe and will leave you in an unrecoverable state if something fails during the merge\n");70    printf("\n");71}72 73// return convert string, for example "128M" or "4G" to number of bytes74static size_t split_str_to_n_bytes(std::string str) {75    size_t n_bytes = 0;76    int n;77    if (str.back() == 'M') {78        sscanf(str.c_str(), "%d", &n);79        n_bytes = (size_t)n * 1000 * 1000; // megabytes80    } else if (str.back() == 'G') {81        sscanf(str.c_str(), "%d", &n);82        n_bytes = (size_t)n * 1000 * 1000 * 1000; // gigabytes83    } else {84        throw std::invalid_argument("error: supported units are M (megabytes) or G (gigabytes), but got: " + std::string(1, str.back()));85    }86    if (n <= 0) {87        throw std::invalid_argument("error: size must be a positive value");88    }89    return n_bytes;90}91 92static void split_params_parse_ex(int argc, const char ** argv, split_params & params) {93    std::string arg;94    const std::string arg_prefix = "--";95    bool invalid_param = false;96 97    int arg_idx = 1;98    for (; arg_idx < argc && strncmp(argv[arg_idx], "--", 2) == 0; arg_idx++) {99        arg = argv[arg_idx];100        if (arg.compare(0, arg_prefix.size(), arg_prefix) == 0) {101            std::replace(arg.begin(), arg.end(), '_', '-');102        }103 104        bool arg_found = false;105        if (arg == "-h" || arg == "--help") {106            split_print_usage(argv[0]);107            exit(0);108        } else if (arg == "--version") {109            fprintf(stderr, "version: %s (build %d, commit %s)\n", llama_version(), llama_build_number(), llama_commit());110            fprintf(stderr, "built with %s for %s\n", llama_compiler(), llama_build_target());111            exit(0);112        } else if (arg == "--dry-run") {113            arg_found = true;114            params.dry_run = true;115        } else if (arg == "--no-tensor-first-split") {116            arg_found = true;117            params.no_tensor_first_split = true;118        } else if (arg == "--merge") {119            arg_found = true;120            if (params.operation != OP_NONE && params.operation != OP_MERGE) {121                throw std::invalid_argument("error: either --split or --merge can be specified, but not both");122            }123            params.operation = OP_MERGE;124        } else if (arg == "--split") {125            arg_found = true;126            if (params.operation != OP_NONE && params.operation != OP_SPLIT) {127                throw std::invalid_argument("error: either --split or --merge can be specified, but not both");128            }129            params.operation = OP_SPLIT;130        } else if (arg == "--split-max-tensors") {131            if (++arg_idx >= argc) {132                invalid_param = true;133                break;134            }135            arg_found = true;136            if (params.mode != MODE_NONE && params.mode != MODE_TENSOR) {137                throw std::invalid_argument("error: either --split-max-tensors or --split-max-size can be specified, but not both");138            }139            params.mode = MODE_TENSOR;140            params.n_split_tensors = atoi(argv[arg_idx]);141        } else if (arg == "--split-max-size") {142            if (++arg_idx >= argc) {143                invalid_param = true;144                break;145            }146            arg_found = true;147            if (params.mode != MODE_NONE && params.mode != MODE_SIZE) {148                throw std::invalid_argument("error: either --split-max-tensors or --split-max-size can be specified, but not both");149            }150            params.mode = MODE_SIZE;151            params.n_bytes_split = split_str_to_n_bytes(argv[arg_idx]);152        } else if (arg == "--delete-splits") {153            arg_found = true;154            params.delete_splits = true;155        }156 157        if (!arg_found) {158            throw std::invalid_argument("error: unknown argument: " + arg);159        }160    }161 162    // the operation is split if not specified163    if (params.operation == OP_NONE) {164        params.operation = OP_SPLIT;165    }166    // the split mode is by tensor if not specified167    if (params.mode == MODE_NONE) {168        params.mode = MODE_TENSOR;169    }170 171    if (invalid_param) {172        throw std::invalid_argument("error: invalid parameter for argument: " + arg);173    }174 175    if (argc - arg_idx != 2) {176        throw std::invalid_argument("error: bad arguments");177    }178 179    params.input = argv[arg_idx++];180    params.output = argv[arg_idx++];181}182 183static bool split_params_parse(int argc, const char ** argv, split_params & params) {184    bool result = true;185    try {186        split_params_parse_ex(argc, argv, params);187    }188    catch (const std::invalid_argument & ex) {189        fprintf(stderr, "%s\n", ex.what());190        split_print_usage(argv[0]);191        exit(EXIT_FAILURE);192    }193    return result;194}195 196static void zeros(std::ofstream & file, size_t n) {197    char zero = 0;198    for (size_t i = 0; i < n; ++i) {199        file.write(&zero, 1);200    }201}202 203struct split_strategy {204    const split_params params;205    std::ifstream & f_input;206    struct gguf_context * ctx_gguf;207    struct ggml_context * ctx_meta = NULL;208    const int n_tensors;209 210    // one ctx_out per one output file211    std::vector<struct gguf_context *> ctx_outs;212 213    // temporary buffer for reading in tensor data214    std::vector<uint8_t> read_buf;215 216    split_strategy(const split_params & params,217            std::ifstream & f_input,218            struct gguf_context * ctx_gguf,219            struct ggml_context * ctx_meta) :220        params(params),221        f_input(f_input),222        ctx_gguf(ctx_gguf),223        ctx_meta(ctx_meta),224        n_tensors(gguf_get_n_tensors(ctx_gguf)) {225 226        // because we need to know list of tensors for each file in advance, we will build all the ctx_out for all output splits227        int i_split = -1;228        struct gguf_context * ctx_out = NULL;229        auto new_ctx_out = [&](bool allow_no_tensors) {230            i_split++;231            if (ctx_out != NULL) {232                if (gguf_get_n_tensors(ctx_out) == 0 && !allow_no_tensors) {233                    fprintf(stderr, "error: one of splits have 0 tensors. Maybe size or tensors limit is too small\n");234                    exit(EXIT_FAILURE);235                }236                ctx_outs.push_back(ctx_out);237            }238            ctx_out = gguf_init_empty();239            // Save all metadata in first split only240            if (i_split == 0) {241                gguf_set_kv(ctx_out, ctx_gguf);242            }243            gguf_set_val_u16(ctx_out, LLM_KV_SPLIT_NO, i_split);244            gguf_set_val_u16(ctx_out, LLM_KV_SPLIT_COUNT, 0); // placeholder245            gguf_set_val_i32(ctx_out, LLM_KV_SPLIT_TENSORS_COUNT, n_tensors);246        };247 248        // initialize ctx_out for the first split249        new_ctx_out(false);250 251        // skip first split if no_tensor_first_split is set252        if (params.no_tensor_first_split) {253            new_ctx_out(true);254        }255 256        // process tensors one by one257        size_t curr_tensors_size = 0; // current size by counting only tensors size (without metadata)258        for (int i = 0; i < n_tensors; ++i) {259            struct ggml_tensor * t = ggml_get_tensor(ctx_meta, gguf_get_tensor_name(ctx_gguf, i));260            // calculate the "imaginary" size = the current size + next tensor size261            size_t n_bytes = GGML_PAD(ggml_nbytes(t), GGUF_DEFAULT_ALIGNMENT);262            size_t next_tensors_size = curr_tensors_size + n_bytes;263            if (should_split(i, next_tensors_size)) {264                new_ctx_out(false);265                curr_tensors_size = n_bytes;266            } else {267                curr_tensors_size = next_tensors_size;268            }269            gguf_add_tensor(ctx_out, t);270        }271 272        // push the last ctx_out273        ctx_outs.push_back(ctx_out);274 275        // set the correct n_split for all ctx_out276        for (auto & ctx : ctx_outs) {277            gguf_set_val_u16(ctx, LLM_KV_SPLIT_COUNT, ctx_outs.size());278        }279    }280 281    ~split_strategy() {282        for (auto & ctx_out : ctx_outs) {283            gguf_free(ctx_out);284        }285    }286 287    bool should_split(int i_tensor, size_t next_size) {288        if (params.mode == MODE_SIZE) {289            // split by max size per file290            return next_size > params.n_bytes_split;291        } else if (params.mode == MODE_TENSOR) {292            // split by number of tensors per file293            return i_tensor > 0 && i_tensor < n_tensors && i_tensor % params.n_split_tensors == 0;294        }295        // should never happen296        GGML_ABORT("invalid mode");297    }298 299    void print_info() {300        printf("n_split: %zu\n", ctx_outs.size());301        int i_split = 0;302        for (auto & ctx_out : ctx_outs) {303            // re-calculate the real gguf size for each split (= metadata size + total size of all tensors)304            size_t total_size = gguf_get_meta_size(ctx_out);305            for (int i = 0; i < gguf_get_n_tensors(ctx_out); ++i) {306                struct ggml_tensor * t = ggml_get_tensor(ctx_meta, gguf_get_tensor_name(ctx_out, i));307                total_size += ggml_nbytes(t);308            }309            total_size = total_size / 1000 / 1000; // convert to megabytes310            printf("split %05d: n_tensors = %" PRIi64 ", total_size = %zuM\n", i_split + 1, gguf_get_n_tensors(ctx_out), total_size);311            i_split++;312        }313    }314 315    void write() {316        int i_split = 0;317        int n_split = ctx_outs.size();318        for (auto & ctx_out : ctx_outs) {319            // construct file path320            char split_path[PATH_MAX] = {0};321            llama_split_path(split_path, sizeof(split_path), params.output.c_str(), i_split, n_split);322 323            // open the output file324            printf("Writing file %s ... ", split_path);325            fflush(stdout);326            std::ofstream fout = std::ofstream(split_path, std::ios::binary);327            fout.exceptions(std::ofstream::failbit); // fail fast on write errors328 329            // write metadata330            std::vector<uint8_t> data(gguf_get_meta_size(ctx_out));331            gguf_get_meta_data(ctx_out, data.data());332            fout.write((const char *)data.data(), data.size());333 334            // write tensors335            for (int i = 0; i < gguf_get_n_tensors(ctx_out); ++i) {336                // read tensor meta and prepare buffer337                const char * t_name = gguf_get_tensor_name(ctx_out, i);338                struct ggml_tensor * t = ggml_get_tensor(ctx_meta, t_name);339                auto n_bytes = ggml_nbytes(t);340                read_buf.resize(n_bytes);341 342                // calculate offset343                auto i_tensor_in = gguf_find_tensor(ctx_gguf, t_name); // idx of tensor in the input file344                auto offset = gguf_get_data_offset(ctx_gguf) + gguf_get_tensor_offset(ctx_gguf, i_tensor_in);345 346                // copy tensor from input to output file347                copy_file_to_file(f_input, fout, offset, n_bytes);348                zeros(fout, GGML_PAD(n_bytes, GGUF_DEFAULT_ALIGNMENT) - n_bytes);349            }350 351            printf("done\n");352            // close the file353            fout.close();354            i_split++;355        }356    }357 358    void copy_file_to_file(std::ifstream & f_in, std::ofstream & f_out, const size_t in_offset, const size_t len) {359        // TODO: detect OS and use copy_file_range() here for better performance360        if (read_buf.size() < len) {361            read_buf.resize(len);362        }363        f_in.seekg(in_offset);364        f_in.read((char *)read_buf.data(), len);365        f_out.write((const char *)read_buf.data(), len);366    }367};368 369static void gguf_split(const split_params & split_params) {370    struct ggml_context * ctx_meta = NULL;371 372    struct gguf_init_params params = {373        /*.no_alloc = */ true,374        /*.ctx      = */ &ctx_meta,375    };376 377    std::ifstream f_input(split_params.input.c_str(), std::ios::binary);378    if (!f_input.is_open()) {379        fprintf(stderr, "%s:  failed to open input GGUF from %s\n", __func__, split_params.input.c_str());380        exit(EXIT_FAILURE);381    }382 383    auto * ctx_gguf = gguf_init_from_file(split_params.input.c_str(), params);384    if (!ctx_gguf) {385        fprintf(stderr, "%s:  failed to load input GGUF from %s\n", __func__, split_params.input.c_str());386        exit(EXIT_FAILURE);387    }388 389    // prepare the strategy390    split_strategy strategy(split_params, f_input, ctx_gguf, ctx_meta);391    int n_split = strategy.ctx_outs.size();392    strategy.print_info();393 394    if (!split_params.dry_run) {395        // write all output splits396        strategy.write();397    }398 399    // done, clean up400    gguf_free(ctx_gguf);401    f_input.close();402 403    fprintf(stderr, "%s: %d gguf split written with a total of %d tensors.\n",404            __func__, n_split, strategy.n_tensors);405}406 407static void gguf_merge(const split_params & split_params) {408    fprintf(stderr, "%s: %s -> %s\n",409            __func__, split_params.input.c_str(),410            split_params.output.c_str());411    int n_split = 1;412    int total_tensors = 0;413 414    // avoid overwriting existing output file415    if (std::ifstream(split_params.output.c_str())) {416        fprintf(stderr, "%s: output file %s already exists\n", __func__, split_params.output.c_str());417        exit(EXIT_FAILURE);418    }419 420 421    auto * ctx_out = gguf_init_empty();422 423    std::vector<uint8_t> read_data;424    std::vector<ggml_context *> ctx_metas;425    std::vector<gguf_context *> ctx_ggufs;426 427    char split_path[PATH_MAX] = {0};428    strncpy(split_path, split_params.input.c_str(), sizeof(split_path) - 1);429    char split_prefix[PATH_MAX] = {0};430 431    // First pass to find KV and tensors metadata432    for (int i_split = 0; i_split < n_split; i_split++) {433        struct ggml_context * ctx_meta = NULL;434 435        struct gguf_init_params params = {436            /*.no_alloc = */ true,437            /*.ctx      = */ &ctx_meta,438        };439 440        if (i_split > 0) {441            llama_split_path(split_path, sizeof(split_path), split_prefix, i_split, n_split);442        }443        fprintf(stderr, "%s: reading metadata %s ...", __func__, split_path);444 445        auto * ctx_gguf = gguf_init_from_file(split_path, params);446        if (!ctx_gguf) {447            fprintf(stderr, "\n%s:  failed to load input GGUF from %s\n", __func__, split_params.input.c_str());448            exit(EXIT_FAILURE);449        }450        ctx_ggufs.push_back(ctx_gguf);451        ctx_metas.push_back(ctx_meta);452 453        if (i_split == 0) {454            auto key_n_split = gguf_find_key(ctx_gguf, LLM_KV_SPLIT_COUNT);455            if (key_n_split < 0) {456                fprintf(stderr,457                        "\n%s: input file does not contain %s metadata\n",458                        __func__,459                        LLM_KV_SPLIT_COUNT);460                gguf_free(ctx_gguf);461                ggml_free(ctx_meta);462                gguf_free(ctx_out);463                exit(EXIT_FAILURE);464            }465 466            n_split = gguf_get_val_u16(ctx_gguf, key_n_split);467            if (n_split < 1) {468                fprintf(stderr,469                        "\n%s: input file does not contain a valid split count %d\n",470                        __func__,471                        n_split);472                gguf_free(ctx_gguf);473                ggml_free(ctx_meta);474                gguf_free(ctx_out);475                exit(EXIT_FAILURE);476            }477 478            // Verify the file naming and extract split_prefix479            if (!llama_split_prefix(split_prefix, sizeof (split_prefix), split_path, i_split, n_split)) {480                fprintf(stderr, "\n%s: unexpected input file name: %s"481                                " i_split=%d"482                                " n_split=%d\n", __func__,483                        split_path, i_split, n_split);484                gguf_free(ctx_gguf);485                ggml_free(ctx_meta);486                gguf_free(ctx_out);487                exit(EXIT_FAILURE);488            }489 490            // Do not trigger merge if we try to merge again the output491            gguf_set_val_u16(ctx_gguf, LLM_KV_SPLIT_COUNT, 0);492 493            // Set metadata from the first split494            gguf_set_kv(ctx_out, ctx_gguf);495        }496 497        auto n_tensors = gguf_get_n_tensors(ctx_gguf);498        for (int i_tensor = 0; i_tensor < n_tensors; i_tensor++) {499            const char * t_name = gguf_get_tensor_name(ctx_gguf, i_tensor);500            struct ggml_tensor * t = ggml_get_tensor(ctx_meta, t_name);501            gguf_add_tensor(ctx_out, t);502        }503        total_tensors += n_tensors;504 505        fprintf(stderr, "\033[3Ddone\n");506    }507    std::ofstream fout;508    if (!split_params.dry_run) {509        fout.open(split_params.output.c_str(), std::ios::binary);510        fout.exceptions(std::ofstream::failbit); // fail fast on write errors511        // placeholder for the meta data512        auto meta_size = gguf_get_meta_size(ctx_out);513        ::zeros(fout, meta_size);514    }515 516    // Write tensors data517    bool merge_error = false;518    for (int i_split = 0; i_split < n_split; i_split++) {519        llama_split_path(split_path, sizeof(split_path), split_prefix, i_split, n_split);520        std::ifstream f_input(split_path, std::ios::binary);521        if (!f_input.is_open()) {522            fprintf(stderr, "%s:  failed to open input GGUF from %s\n", __func__, split_path);523            for (uint32_t i = 0; i < ctx_ggufs.size(); i++) {524                gguf_free(ctx_ggufs[i]);525                ggml_free(ctx_metas[i]);526            }527            gguf_free(ctx_out);528            if (!split_params.dry_run) {529                fout.close();530            }531            exit(EXIT_FAILURE);532        }533        fprintf(stderr, "%s: writing tensors %s ...", __func__, split_path);534 535        auto * ctx_gguf = ctx_ggufs[i_split];536        auto * ctx_meta = ctx_metas[i_split];537 538        auto n_tensors = gguf_get_n_tensors(ctx_gguf);539        for (int i_tensor = 0; i_tensor < n_tensors; i_tensor++) {540            const char * t_name = gguf_get_tensor_name(ctx_gguf, i_tensor);541            struct ggml_tensor * t = ggml_get_tensor(ctx_meta, t_name);542 543            auto n_bytes = ggml_nbytes(t);544 545            if (read_data.size() < n_bytes) {546                read_data.resize(n_bytes);547            }548 549            auto offset = gguf_get_data_offset(ctx_gguf) + gguf_get_tensor_offset(ctx_gguf, i_tensor);550            f_input.seekg(offset);551            f_input.read((char *)read_data.data(), n_bytes);552            if (!split_params.dry_run) {553                // write tensor data + padding554                fout.write((const char *)read_data.data(), n_bytes);555                zeros(fout, GGML_PAD(n_bytes, GGUF_DEFAULT_ALIGNMENT) - n_bytes);556            }557        }558 559        gguf_free(ctx_gguf);560        ggml_free(ctx_meta);561        f_input.close();562        fprintf(stderr, "\033[3Ddone\n");563 564        if (!split_params.dry_run && split_params.delete_splits) {565            int delete_result = std::remove(split_path);566            if (delete_result != 0) {567                merge_error = true;568                fprintf(stderr, "error: failed to delete %s\n", split_path);569            } else {570                fprintf(stderr, "%s: deleted file %s\n", __func__, split_path);571            }572        }573    }574 575    if (!split_params.dry_run) {576        // go back to beginning of file and write the updated metadata577        fout.seekp(0);578        std::vector<uint8_t> data(gguf_get_meta_size(ctx_out));579        gguf_get_meta_data(ctx_out, data.data());580        fout.write((const char *)data.data(), data.size());581        fout.close();582    }583    gguf_free(ctx_out);584 585    fprintf(stderr, "%s: %s merged from %d split with %d tensors.\n",586            __func__, split_params.output.c_str(), n_split, total_tensors);587 588    if (merge_error) {589        exit(EXIT_FAILURE);590    }591}592 593int main(int argc, const char ** argv) {594    std::setlocale(LC_NUMERIC, "C");595 596    split_params params;597    split_params_parse(argc, argv, params);598 599    switch (params.operation) {600        case OP_SPLIT: gguf_split(params);601            break;602        case OP_MERGE: gguf_merge(params);603            break;604        default: split_print_usage(argv[0]);605            exit(EXIT_FAILURE);606    }607 608    return 0;609}610