Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
llama-model-loader.h212 linesDownload Raw Back to src
1#pragma once2 3#include "llama.h"4 5#include "llama-impl.h"6#include "llama-arch.h"7#include "llama-hparams.h"8#include "llama-mmap.h"9 10#include "ggml-cpp.h"11 12#include <cstddef>13#include <cstring>14#include <map>15#include <stdexcept>16#include <unordered_map>17 18using llama_buf_map = std::unordered_map<uint32_t, ggml_backend_buffer_t>;19 20// lists of buffer types used for each layer21using buft_list_t = std::vector<std::pair<ggml_backend_dev_t, ggml_backend_buffer_type_t>>;22 23enum llama_fver {24    GGUF_FILE_VERSION_V1 = 1,25    GGUF_FILE_VERSION_V2 = 2,26    GGUF_FILE_VERSION_V3 = 3,27};28 29const char * llama_file_version_name(llama_fver version);30 31struct llama_model_loader {32    // Holds information on a model weight33    struct llama_tensor_weight {34        uint16_t  idx; // source file index35        size_t   offs; // tensor data offset in the original file36 37        ggml_tensor * tensor;38 39        llama_tensor_weight(const llama_file * file, uint16_t idx, const struct gguf_context * gguf_ctx, ggml_tensor * tensor) : idx(idx), tensor(tensor) {40            const int tensor_idx = gguf_find_tensor(gguf_ctx,  ggml_get_name(tensor));41            if (tensor_idx < 0) {42                throw std::runtime_error(format("tensor '%s' not found in the model", ggml_get_name(tensor)));43            }44 45            offs = gguf_get_data_offset(gguf_ctx) + gguf_get_tensor_offset(gguf_ctx, tensor_idx);46            if (offs + ggml_nbytes(tensor) < offs || offs + ggml_nbytes(tensor) > file->size()) {47                throw std::runtime_error(format("tensor '%s' data is not within the file bounds, model is corrupted or incomplete", ggml_get_name(tensor)));48            }49        }50    };51 52    // custom comparator to sort weights more nicely by layer53    struct weight_name_comparer {54        bool operator()(const std::string & a, const std::string & b) const {55            int a_layer = -1;56            int b_layer = -1;57            sscanf(a.c_str(), "blk.%d.", &a_layer);58            sscanf(b.c_str(), "blk.%d.", &b_layer);59            if (a_layer != b_layer) {60                return a_layer < b_layer;61            }62            return a < b;63        }64    };65 66    static const int TENSOR_NOT_REQUIRED    = 1 << 0;67    static const int TENSOR_DUPLICATED      = 1 << 1;68    static const int TENSOR_SKIP            = 1 << 2;69    static const int TENSOR_SKIP_IF_VIRTUAL = 1 << 3;70    static const int TENSOR_ALLOW_RESHAPE   = 1 << 4;71 72    int n_kv      = 0;73    int n_tensors = 0;74    int n_created = 0;75 76    uint64_t n_elements = 0;77    size_t   n_bytes    = 0;78 79    bool use_mmap = false;80    bool use_direct_io = false;81    bool check_tensors;82    bool no_alloc;83    bool load_mtp;84 85    llama_files files;86    llama_ftype ftype;87    llama_fver  fver;88 89    llama_mmaps mappings;90 91    std::map<std::string, llama_tensor_weight, weight_name_comparer> weights_map;92    std::unordered_map<std::string, llama_model_kv_override> kv_overrides;93    const llama_model_tensor_buft_override * tensor_buft_overrides;94 95    gguf_context_ptr metadata_ptr;96    struct gguf_context * metadata; // either metadata_ptr.get() or externally set97    llama_model_set_tensor_data_t set_tensor_data;98    void * set_tensor_data_ud;99    std::vector<ggml_context_ptr> contexts;100 101    std::string arch_name;102    LLM_KV      llm_kv    = LLM_KV(LLM_ARCH_UNKNOWN);103 104    size_t size_done = 0;105    size_t size_data = 0;106    std::vector<std::pair<size_t, size_t>> mmaps_used;107 108    // define a comparator for the buft -> ctx map to ensure that the order is well-defined:109    struct ggml_backend_buft_comparator {110        bool operator()(const ggml_backend_buffer_type_t & lhs, const ggml_backend_buffer_type_t & rhs) const {111            return strcmp(ggml_backend_buft_name(lhs), ggml_backend_buft_name(rhs)) < 0;112        }113    };114 115    std::map<ggml_backend_buffer_type_t, ggml_context_ptr, ggml_backend_buft_comparator> ctx_map;116 117    // track tensors that had to be moved for debugging:118    size_t n_tensors_moved = 0;119    std::string first_tensor_moved_name;120    std::string first_tensor_moved_type_name;121    ggml_backend_buffer_type_t first_moved_from_buft = nullptr;122    ggml_backend_buffer_type_t first_moved_to_buft = nullptr;123 124    llama_model_loader(125        struct gguf_context * metadata,126        llama_model_set_tensor_data_t set_tensor_data,127        void * set_tensor_data_ud,128        const std::string & fname,129        std::vector<std::string> & splits, // optional, only need if the split does not follow naming scheme130        FILE * file,131        llama_load_mode load_mode,132        bool check_tensors,133        bool no_alloc,134        bool load_mtp,135        const llama_model_kv_override * param_overrides_p,136        const llama_model_tensor_buft_override * param_tensor_buft_overrides_p);137 138    template<typename T>139    typename std::enable_if<std::is_integral<T>::value, bool>::type140    get_arr_n(const std::string & key, T & result, bool required = true);141 142    template<typename T>143    typename std::enable_if<std::is_integral<T>::value, bool>::type144    get_arr_n(enum llm_kv kid, T & result, bool required = true);145 146    template<typename T>147    bool get_arr(const std::string & key, std::vector<T> & result, bool required = true);148 149    template<typename T, size_t N_MAX>150    bool get_arr(const std::string & key, std::array<T, N_MAX> & result, bool required = true);151 152    template<typename T>153    bool get_arr(enum llm_kv kid, T & result, bool required = true);154 155    template<typename T>156    bool get_key(const std::string & key, T & result, bool required = true);157 158    template<typename T>159    bool get_key(enum llm_kv kid, T & result, bool required = true);160 161    template<typename T, size_t N_MAX>162    bool get_key_or_arr(const std::string & key, std::array<T, N_MAX> & result, uint32_t n, bool required = true);163 164    template<typename T>165    bool get_key_or_arr(enum llm_kv kid, T & result, uint32_t n, bool required = true);166 167    bool get_key_or_arr(enum llm_kv kid, uint32_t & result, bool required = true);168 169    std::string get_arch_name() const;170 171    enum llm_arch get_arch() const;172 173    const llama_tensor_weight * get_weight(const char * name) const;174 175    const llama_tensor_weight & require_weight(const char * name) const;176 177    struct ggml_tensor * get_tensor_meta(const char * name) const;178 179    struct ggml_tensor * require_tensor_meta(const std::string & name) const;180 181    const struct ggml_tensor * check_tensor_dims(182            const std::string & name,183            const std::vector<int64_t> & ne,184            bool required,185            bool allow_reshape) const;186 187    struct ggml_tensor * create_tensor(188        const llama_hparams & hparams, const buft_list_t * buft_list_cpu, const buft_list_t * buft_list_input, const buft_list_t * buft_list_output,189        const buft_list_t * buft_list_layer, const LLM_TN_IMPL & tn, const std::initializer_list<int64_t> & ne, int flags);190 191    void done_getting_tensors(bool partial = false) const;192 193    void init_mappings(bool prefetch = true, llama_mlocks * mlock_mmaps = nullptr);194 195    void get_mapping_range(size_t * first, size_t * last, void ** addr, int idx, ggml_context * ctx) const;196 197    // for backwards compatibility, does not support ggml-backend198    void load_data_for(struct ggml_tensor * cur) const;199 200    // Returns false if cancelled by progress_callback201    bool load_all_data(202            struct ggml_context * ctx,203            llama_buf_map & bufs,204            llama_mlocks * lmlocks,205            llama_progress_callback progress_callback,206            void * progress_callback_user_data);207 208    std::string ftype_name() const;209 210    void print_info() const;211};212 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai