Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
llama-model-loader.h168 linesDownload Raw Back to src
1#pragma once2 3#include "llama.h"4 5#include "llama-impl.h"6#include "llama-arch.h"7#include "llama-mmap.h"8 9#include "ggml-cpp.h"10 11#include <cstddef>12#include <map>13#include <stdexcept>14#include <unordered_map>15 16using llama_buf_map = std::unordered_map<uint32_t, ggml_backend_buffer_t>;17 18enum llama_fver {19    GGUF_FILE_VERSION_V1 = 1,20    GGUF_FILE_VERSION_V2 = 2,21    GGUF_FILE_VERSION_V3 = 3,22};23 24const char * llama_file_version_name(llama_fver version);25 26struct llama_model_loader {27    // Holds information on a model weight28    struct llama_tensor_weight {29        uint16_t  idx; // source file index30        size_t   offs; // tensor data offset in the original file31 32        ggml_tensor * tensor;33 34        llama_tensor_weight(const llama_file * file, uint16_t idx, const struct gguf_context * gguf_ctx, ggml_tensor * tensor) : idx(idx), tensor(tensor) {35            const int tensor_idx = gguf_find_tensor(gguf_ctx,  ggml_get_name(tensor));36            if (tensor_idx < 0) {37                throw std::runtime_error(format("tensor '%s' not found in the model", ggml_get_name(tensor)));38            }39 40            offs = gguf_get_data_offset(gguf_ctx) + gguf_get_tensor_offset(gguf_ctx, tensor_idx);41            if (offs + ggml_nbytes(tensor) < offs || offs + ggml_nbytes(tensor) > file->size()) {42                throw std::runtime_error(format("tensor '%s' data is not within the file bounds, model is corrupted or incomplete", ggml_get_name(tensor)));43            }44        }45    };46 47    // custom comparator to sort weights more nicely by layer48    struct weight_name_comparer {49        bool operator()(const std::string & a, const std::string & b) const {50            int a_layer = -1;51            int b_layer = -1;52            sscanf(a.c_str(), "blk.%d.", &a_layer);53            sscanf(b.c_str(), "blk.%d.", &b_layer);54            if (a_layer != b_layer) {55                return a_layer < b_layer;56            }57            return a < b;58        }59    };60 61    static const int TENSOR_NOT_REQUIRED = 1;62    static const int TENSOR_DUPLICATED   = 2;63 64    int n_kv      = 0;65    int n_tensors = 0;66    int n_created = 0;67 68    uint64_t n_elements = 0;69    size_t   n_bytes    = 0;70 71    bool use_mmap = false;72    bool check_tensors;73 74    llama_files files;75    llama_ftype ftype;76    llama_fver  fver;77 78    llama_mmaps mappings;79 80    std::map<std::string, struct llama_tensor_weight, weight_name_comparer> weights_map;81    std::unordered_map<std::string, struct llama_model_kv_override> kv_overrides;82 83    gguf_context_ptr meta;84    std::vector<ggml_context_ptr> contexts;85 86    std::string arch_name;87    LLM_KV      llm_kv    = LLM_KV(LLM_ARCH_UNKNOWN);88 89    size_t size_done = 0;90    size_t size_data = 0;91    std::vector<std::pair<size_t, size_t>> mmaps_used;92 93    llama_model_loader(94        const std::string & fname,95        std::vector<std::string> & splits, // optional, only need if the split does not follow naming scheme96        bool use_mmap,97        bool check_tensors,98        const struct llama_model_kv_override * param_overrides_p);99 100    template<typename T>101    typename std::enable_if<std::is_integral<T>::value, bool>::type102    get_arr_n(const std::string & key, T & result, bool required = true);103 104    template<typename T>105    typename std::enable_if<std::is_integral<T>::value, bool>::type106    get_arr_n(enum llm_kv kid, T & result, bool required = true);107 108    template<typename T>109    bool get_arr(const std::string & key, std::vector<T> & result, bool required = true);110 111    template<typename T, size_t N_MAX>112    bool get_arr(const std::string & key, std::array<T, N_MAX> & result, bool required = true);113 114    template<typename T>115    bool get_arr(enum llm_kv kid, T & result, bool required = true);116 117    template<typename T>118    bool get_key(const std::string & key, T & result, bool required = true);119 120    template<typename T>121    bool get_key(enum llm_kv kid, T & result, bool required = true);122 123    template<typename T, size_t N_MAX>124    bool get_key_or_arr(const std::string & key, std::array<T, N_MAX> & result, uint32_t n, bool required = true);125 126    template<typename T>127    bool get_key_or_arr(enum llm_kv kid, T & result, uint32_t n, bool required = true);128 129    std::string get_arch_name() const;130 131    enum llm_arch get_arch() const;132 133    const llama_tensor_weight * get_weight(const char * name) const;134 135    const llama_tensor_weight & require_weight(const char * name) const;136 137    struct ggml_tensor * get_tensor_meta(const char * name) const;138 139    struct ggml_tensor * require_tensor_meta(const std::string & name) const;140 141    const struct ggml_tensor * check_tensor_dims(const std::string & name, const std::vector<int64_t> & ne, bool required) const;142 143    struct ggml_tensor * create_tensor(struct ggml_context * ctx, const std::string & name, const std::initializer_list<int64_t> & ne, int flags = 0);144 145    struct ggml_tensor * create_tensor_as_view(struct ggml_context * ctx, struct ggml_tensor * base, const std::string & name, const std::initializer_list<int64_t> & ne, size_t offset, bool required = true);146 147    void done_getting_tensors() const;148 149    void init_mappings(bool prefetch = true, llama_mlocks * mlock_mmaps = nullptr);150 151    void get_mapping_range(size_t * first, size_t * last, void ** addr, int idx, ggml_context * ctx) const;152 153    // for backwards compatibility, does not support ggml-backend154    void load_data_for(struct ggml_tensor * cur) const;155 156    // Returns false if cancelled by progress_callback157    bool load_all_data(158            struct ggml_context * ctx,159            llama_buf_map & bufs,160            llama_mlocks * lmlocks,161            llama_progress_callback progress_callback,162            void * progress_callback_user_data);163 164    std::string ftype_name() const;165 166    void print_info() const;167};168