Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
gguf.cpp1698 linesDownload Raw Back to src
1#include "ggml.h"2#include "ggml-backend.h"3#include "ggml-impl.h"4#include "gguf.h"5 6#include <cinttypes>7#include <cstddef>8#include <cstdint>9#include <cstdio>10#include <cstdlib>11#include <cstring>12#include <map>13#include <new>14#include <stdexcept>15#include <string>16#include <vector>17 18#define GGUF_MAX_STRING_LENGTH  (1024*1024*1024)19#define GGUF_MAX_ARRAY_ELEMENTS (1024*1024*1024)20 21#ifdef _WIN3222#    define gguf_ftell _ftelli6423#    define gguf_fseek _fseeki6424#else25#    define gguf_ftell ftello26#    define gguf_fseek fseeko27#endif28 29template <typename T>30struct type_to_gguf_type;31 32template <>33struct type_to_gguf_type<uint8_t> {34    static constexpr enum gguf_type value = GGUF_TYPE_UINT8;35};36 37template <>38struct type_to_gguf_type<int8_t> {39    static constexpr enum gguf_type value = GGUF_TYPE_INT8;40};41 42template <>43struct type_to_gguf_type<uint16_t> {44    static constexpr enum gguf_type value = GGUF_TYPE_UINT16;45};46 47template <>48struct type_to_gguf_type<int16_t> {49    static constexpr enum gguf_type value = GGUF_TYPE_INT16;50};51 52template <>53struct type_to_gguf_type<uint32_t> {54    static constexpr enum gguf_type value = GGUF_TYPE_UINT32;55};56 57template <>58struct type_to_gguf_type<int32_t> {59    static constexpr enum gguf_type value = GGUF_TYPE_INT32;60};61 62template <>63struct type_to_gguf_type<float> {64    static constexpr enum gguf_type value = GGUF_TYPE_FLOAT32;65};66 67template <>68struct type_to_gguf_type<bool> {69    static constexpr enum gguf_type value = GGUF_TYPE_BOOL;70};71 72template <>73struct type_to_gguf_type<std::string> {74    static constexpr enum gguf_type value = GGUF_TYPE_STRING;75};76 77template <>78struct type_to_gguf_type<uint64_t> {79    static constexpr enum gguf_type value = GGUF_TYPE_UINT64;80};81 82template <>83struct type_to_gguf_type<int64_t> {84    static constexpr enum gguf_type value = GGUF_TYPE_INT64;85};86 87template <>88struct type_to_gguf_type<double> {89    static constexpr enum gguf_type value = GGUF_TYPE_FLOAT64;90};91 92static const std::map<gguf_type, size_t> GGUF_TYPE_SIZE = {93    {GGUF_TYPE_UINT8,   sizeof(uint8_t)},94    {GGUF_TYPE_INT8,    sizeof(int8_t)},95    {GGUF_TYPE_UINT16,  sizeof(uint16_t)},96    {GGUF_TYPE_INT16,   sizeof(int16_t)},97    {GGUF_TYPE_UINT32,  sizeof(uint32_t)},98    {GGUF_TYPE_INT32,   sizeof(int32_t)},99    {GGUF_TYPE_FLOAT32, sizeof(float)},100    {GGUF_TYPE_BOOL,    sizeof(int8_t)},101    {GGUF_TYPE_STRING,  0}, // undefined102    {GGUF_TYPE_ARRAY,   0}, // undefined103    {GGUF_TYPE_UINT64,  sizeof(uint64_t)},104    {GGUF_TYPE_INT64,   sizeof(int64_t)},105    {GGUF_TYPE_FLOAT64, sizeof(double)},106};107static_assert(GGUF_TYPE_COUNT == 13, "GGUF_TYPE_COUNT != 13");108 109static const std::map<gguf_type, const char *> GGUF_TYPE_NAME = {110    {GGUF_TYPE_UINT8,   "u8"},111    {GGUF_TYPE_INT8,    "i8"},112    {GGUF_TYPE_UINT16,  "u16"},113    {GGUF_TYPE_INT16,   "i16"},114    {GGUF_TYPE_UINT32,  "u32"},115    {GGUF_TYPE_INT32,   "i32"},116    {GGUF_TYPE_FLOAT32, "f32"},117    {GGUF_TYPE_BOOL,    "bool"},118    {GGUF_TYPE_STRING,  "str"},119    {GGUF_TYPE_ARRAY,   "arr"},120    {GGUF_TYPE_UINT64,  "u64"},121    {GGUF_TYPE_INT64,   "i64"},122    {GGUF_TYPE_FLOAT64, "f64"},123};124static_assert(GGUF_TYPE_COUNT == 13, "GGUF_TYPE_COUNT != 13");125 126size_t gguf_type_size(enum gguf_type type) {127    auto it = GGUF_TYPE_SIZE.find(type);128    return it == GGUF_TYPE_SIZE.end() ? 0 : it->second;129}130 131struct gguf_kv {132    std::string key;133 134    bool is_array;135    enum gguf_type type;136 137    std::vector<int8_t>      data;138    std::vector<std::string> data_string;139 140    template <typename T>141    gguf_kv(const std::string & key, const T value)142            : key(key), is_array(false), type(type_to_gguf_type<T>::value) {143        GGML_ASSERT(!key.empty());144        data.resize(sizeof(T));145        memcpy(data.data(), &value, sizeof(T));146    }147 148    template <typename T>149    gguf_kv(const std::string & key, const std::vector<T> & value)150            : key(key), is_array(true), type(type_to_gguf_type<T>::value) {151        GGML_ASSERT(!key.empty());152        data.resize(value.size()*sizeof(T));153        for (size_t i = 0; i < value.size(); ++i) {154            const T tmp = value[i];155            memcpy(data.data() + i*sizeof(T), &tmp, sizeof(T));156        }157    }158 159    gguf_kv(const std::string & key, const std::string & value)160            : key(key), is_array(false), type(GGUF_TYPE_STRING) {161        GGML_ASSERT(!key.empty());162        data_string.push_back(value);163    }164 165    gguf_kv(const std::string & key, const std::vector<std::string> & value)166            : key(key), is_array(true), type(GGUF_TYPE_STRING) {167        GGML_ASSERT(!key.empty());168        data_string = value;169    }170 171    const std::string & get_key() const {172        return key;173    }174 175    const enum gguf_type & get_type() const {176        return type;177    }178 179    size_t get_ne() const {180        if (type == GGUF_TYPE_STRING) {181            const size_t ne = data_string.size();182            GGML_ASSERT(is_array || ne == 1);183            return ne;184        }185        const size_t type_size = gguf_type_size(type);186        GGML_ASSERT(data.size() % type_size == 0);187        const size_t ne = data.size() / type_size;188        GGML_ASSERT(is_array || ne == 1);189        return ne;190    }191 192    template <typename T>193    const T & get_val(const size_t i = 0) const {194        GGML_ASSERT(type_to_gguf_type<T>::value == type);195        if constexpr (std::is_same<T, std::string>::value) {196            GGML_ASSERT(data_string.size() >= i+1);197            return data_string[i];198        }199        const size_t type_size = gguf_type_size(type);200        GGML_ASSERT(data.size() % type_size == 0);201        GGML_ASSERT(data.size() >= (i+1)*type_size);202        return reinterpret_cast<const T *>(data.data())[i];203    }204 205    void cast(const enum gguf_type new_type) {206        const size_t new_type_size = gguf_type_size(new_type);207        GGML_ASSERT(data.size() % new_type_size == 0);208        type = new_type;209    }210};211 212struct gguf_tensor_info {213    struct ggml_tensor t; // for holding the equivalent info214    uint64_t offset;      // offset from start of `data`, must be a multiple of `ALIGNMENT`215};216 217struct gguf_context {218    uint32_t version = GGUF_VERSION;219 220    std::vector<struct gguf_kv> kv;221    std::vector<struct gguf_tensor_info> info;222 223    size_t alignment = GGUF_DEFAULT_ALIGNMENT;224    size_t offset    = 0; // offset of `data` from beginning of file225    size_t size      = 0; // size of `data` in bytes226 227    void * data = nullptr;228};229 230struct gguf_reader {231    gguf_reader(232            gguf_reader_callback_t callback,233            void * userdata,234            size_t max_chunk_read,235            uint64_t data_offset = 0,236            uint64_t nbytes_remain = 0)237        : callback(callback),238          userdata(userdata),239          max_chunk_read(max_chunk_read),240          data_offset(data_offset),241          nbytes_remain(nbytes_remain) {242        GGML_ASSERT(max_chunk_read > 0);243    }244 245    // helper for remaining bytes in a file246    static uint64_t file_remain(FILE * file) {247        const int64_t cur = gguf_ftell(file);248        if (cur < 0) {249            return 0;250        }251        if (gguf_fseek(file, 0, SEEK_END) != 0) {252            gguf_fseek(file, cur, SEEK_SET);253 254            return 0;255        }256        const int64_t end = gguf_ftell(file);257        if (end < 0) {258            gguf_fseek(file, cur, SEEK_SET);259 260            return 0;261        }262        gguf_fseek(file, cur, SEEK_SET);263        return static_cast<uint64_t>(end - cur);264    }265 266    template <typename T>267    bool read(T & dst) const {268        const size_t size = sizeof(dst);269        if (size > nbytes_remain) {270            return false;271        }272        return read_raw(&dst, size) == size;273    }274 275    template <typename T>276    bool read(std::vector<T> & dst, const size_t n) const {277        if (n > GGUF_MAX_ARRAY_ELEMENTS) {278            return false;279        }280        if constexpr (std::is_same<T, std::string>::value) {281            // strings are prefixed with their length, so we need to account for that282            if (n > SIZE_MAX / sizeof(uint64_t)) {283                return false;284            }285            if (nbytes_remain < n * sizeof(uint64_t)) {286                return false;287            }288        } else {289            if (n > SIZE_MAX / sizeof(T)) {290                return false;291            }292            if (nbytes_remain < n * sizeof(T)) {293                return false;294            }295        }296        dst.resize(n);297        for (size_t i = 0; i < dst.size(); ++i) {298            if constexpr (std::is_same<T, bool>::value) {299                bool tmp;300                if (!read(tmp)) {301                    return false;302                }303                dst[i] = tmp;304            } else {305                if (!read(dst[i])) {306                    return false;307                }308            }309        }310        return true;311    }312 313    bool read(bool & dst) const {314        int8_t tmp = -1;315        if (!read(tmp)) {316            return false;317        }318        dst = tmp != 0;319        return true;320    }321 322    bool read(enum ggml_type & dst) const {323        int32_t tmp = -1;324        if (!read(tmp)) {325            return false;326        }327        dst = ggml_type(tmp);328        return true;329    }330 331    bool read(enum gguf_type & dst) const {332        int32_t tmp = -1;333        if (!read(tmp)) {334            return false;335        }336        dst = gguf_type(tmp);337        return true;338    }339 340    bool read(std::string & dst) const {341        uint64_t size = 0;342        if (!read(size)) {343            return false;344        }345        if (size > GGUF_MAX_STRING_LENGTH) {346            GGML_LOG_ERROR("%s: string length %" PRIu64 " exceeds maximum %" PRIu64 "\n", __func__, size, (uint64_t) GGUF_MAX_STRING_LENGTH);347            return false;348        }349        if (size > nbytes_remain) {350            GGML_LOG_ERROR("%s: string length %" PRIu64 " exceeds remaining file size %" PRIu64 " bytes\n", __func__, size, nbytes_remain);351            return false;352        }353        dst.resize(static_cast<size_t>(size));354        return read_raw(dst.data(), static_cast<size_t>(size)) == size;355    }356 357    bool read(void * dst, const size_t size) const {358        if (size > nbytes_remain) {359            return false;360        }361        return read_raw(dst, size) == size;362    }363 364    uint64_t tell() const {365        return data_offset;366    }367 368    bool seek(uint64_t absolute_offset) const {369        const uint64_t end_offset = uint64_t(data_offset) + nbytes_remain;370        if (absolute_offset > end_offset) {371            return false;372        }373 374        data_offset = absolute_offset;375        nbytes_remain = end_offset - absolute_offset;376 377        return true;378    }379 380private:381    size_t read_raw(void * dst, size_t size) const {382        if (callback == nullptr || size == 0) {383            return 0;384        }385 386        uint8_t * data = static_cast<uint8_t *>(dst);387        size_t total_nread = 0;388        bool reached_eof = false;389 390        while (total_nread < size) {391            const size_t chunk_size = std::min(max_chunk_read, size - total_nread);392            if (data_offset + total_nread < data_offset) {393                break;394            }395            const size_t nread = callback(userdata, static_cast<void *>(data + total_nread), data_offset + total_nread, chunk_size);396            total_nread += nread;397            if (nread != chunk_size) {398                reached_eof = true;399                break;400            }401        }402 403        data_offset += total_nread;404        GGML_ASSERT(total_nread <= nbytes_remain);405        nbytes_remain -= total_nread;406 407        if (reached_eof) {408            nbytes_remain = 0;409        }410 411        return total_nread;412    }413 414    gguf_reader_callback_t callback = nullptr;415    void * userdata = nullptr;416    size_t max_chunk_read = 0;417    mutable uint64_t data_offset = 0;418    mutable uint64_t nbytes_remain = 0;419};420 421struct gguf_context * gguf_init_empty(void) {422    return new gguf_context;423}424 425template<typename T>426bool gguf_read_emplace_helper(const struct gguf_reader & gr, std::vector<struct gguf_kv> & kv, const std::string & key, const bool is_array, const size_t n) {427    if (is_array) {428        std::vector<T> value;429        try {430            if (!gr.read(value, n)) {431                return false;432            }433        } catch (std::length_error &) {434            GGML_LOG_ERROR("%s: encountered length_error while reading value for key '%s'\n", __func__, key.c_str());435            return false;436        } catch (std::bad_alloc &) {437            GGML_LOG_ERROR("%s: encountered bad_alloc error while reading value for key '%s'\n", __func__, key.c_str());438            return false;439        }440        kv.emplace_back(key, value);441    } else {442        T value;443        if (!gr.read(value)) {444            return false;445        }446        kv.emplace_back(key, value);447    }448    return true;449}450 451static struct gguf_context * gguf_init_from_reader(const struct gguf_reader & gr, struct gguf_init_params params) {452    struct gguf_context * ctx = new gguf_context;453 454    bool ok = true;455 456    // file magic457    {458        std::vector<char> magic;459        ok = ok && gr.read(magic, 4);460 461        if (!ok) {462            GGML_LOG_ERROR("%s: failed to read magic\n", __func__);463            gguf_free(ctx);464            return nullptr;465        }466 467        for (uint32_t i = 0; i < magic.size(); i++) {468            if (magic[i] != GGUF_MAGIC[i]) {469                char c0 = isprint(magic[0]) ? magic[0] : '?';470                char c1 = isprint(magic[1]) ? magic[1] : '?';471                char c2 = isprint(magic[2]) ? magic[2] : '?';472                char c3 = isprint(magic[3]) ? magic[3] : '?';473                GGML_LOG_ERROR("%s: invalid magic characters: '%c%c%c%c', expected 'GGUF'\n", __func__, c0, c1, c2, c3);474                gguf_free(ctx);475                return nullptr;476            }477        }478    }479 480    // header481    int64_t n_kv      = 0;482    int64_t n_tensors = 0;483 484    if (ok && gr.read(ctx->version)) {485        if (ok && ctx->version == 0) {486            GGML_LOG_ERROR("%s: bad GGUF version: %" PRIu32 "\n", __func__, ctx->version);487            ok = false;488        }489 490        /*491         * bit layout is different when reading non-native endian models.492         * assuming that the GGUF version is 3, the non-native endian model493         * would read it as 0x30000000. we can use the AND operation against494         * the last 4 hexadecimal digits to check if the model is the same495         * endianness as the host system.496        */497        if (ok && (ctx->version & 0x0000FFFF) == 0x00000000) {498            GGML_LOG_ERROR("%s: failed to load model: this GGUF file version %" PRIu32 " is extremely large, is there a mismatch between the host and model endianness?\n", __func__, ctx->version);499            ok = false;500        }501 502        if (ok && ctx->version == 1) {503            GGML_LOG_ERROR("%s: GGUFv1 is no longer supported, please use a more up-to-date version\n", __func__);504            ok = false;505        }506        if (ok && ctx->version > GGUF_VERSION) {507            GGML_LOG_ERROR("%s: this GGUF file is version %" PRIu32 " but this software only supports up to version %d\n",508                __func__, ctx->version, GGUF_VERSION);509            ok = false;510        }511    } else {512        ok = false;513    }514 515    if (ok && gr.read(n_tensors)) {516        static_assert(sizeof(size_t) <= 8 && sizeof(gguf_tensor_info) >= 2, "int64_t insufficient for indexing");517        if (n_tensors < 0 || n_tensors > int64_t(SIZE_MAX/sizeof(gguf_tensor_info))) {518            GGML_LOG_ERROR("%s: number of tensors is %" PRIi64 " but must be in [0, %zu]\n",519                __func__, n_tensors, SIZE_MAX/sizeof(gguf_tensor_info));520            ok = false;521        }522    } else {523        ok = false;524    }525 526    if (ok && gr.read(n_kv)) {527        static_assert(sizeof(size_t) <= 8 && sizeof(gguf_tensor_info) >= 2, "int64_t insufficient for indexing");528        if (n_kv < 0 || n_kv > int64_t(SIZE_MAX/sizeof(gguf_kv))) {529            GGML_LOG_ERROR("%s: number of key value pairs is %" PRIi64 " but must be in [0, %zu]\n",530                    __func__, n_kv, SIZE_MAX/sizeof(gguf_kv));531            ok = false;532        }533    } else {534        ok = false;535    }536 537    if (!ok) {538        GGML_LOG_ERROR("%s: failed to read header\n", __func__);539        gguf_free(ctx);540        return nullptr;541    }542 543    // KV pairs544    {545        for (int64_t i = 0; ok && i < n_kv; ++i) {546            std::string key;547            gguf_type   type     = gguf_type(-1);548            bool        is_array = false;549            uint64_t    n        = 1;550 551            try {552                ok = ok && gr.read(key);553            } catch (std::length_error &) {554                GGML_LOG_ERROR("%s: encountered length_error while reading key %" PRIi64 "\n", __func__, i);555                ok = false;556            } catch (std::bad_alloc &) {557                GGML_LOG_ERROR("%s: encountered bad_alloc error while reading key %" PRIi64 "\n", __func__, i);558                ok = false;559            }560            if (ok && key.empty()) {561                GGML_LOG_ERROR("%s: key %" PRIi64 " is empty\n", __func__, i);562                ok = false;563            }564            for (size_t j = 0; ok && j < ctx->kv.size(); ++j) {565                if (key == ctx->kv[j].key) {566                    GGML_LOG_ERROR("%s: duplicate key '%s' for tensors %zu and %" PRIi64 " \n", __func__, key.c_str(), j, i);567                    ok = false;568                }569            }570            if (!ok) {571                break;572            }573 574            ok = ok && gr.read(type);575            if (type == GGUF_TYPE_ARRAY) {576                is_array = true;577                ok = ok && gr.read(type);578                ok = ok && gr.read(n);579            }580            if (!ok) {581                break;582            }583 584            switch (type) {585                case GGUF_TYPE_UINT8:   ok = ok && gguf_read_emplace_helper<uint8_t>    (gr, ctx->kv, key, is_array, n); break;586                case GGUF_TYPE_INT8:    ok = ok && gguf_read_emplace_helper<int8_t>     (gr, ctx->kv, key, is_array, n); break;587                case GGUF_TYPE_UINT16:  ok = ok && gguf_read_emplace_helper<uint16_t>   (gr, ctx->kv, key, is_array, n); break;588                case GGUF_TYPE_INT16:   ok = ok && gguf_read_emplace_helper<int16_t>    (gr, ctx->kv, key, is_array, n); break;589                case GGUF_TYPE_UINT32:  ok = ok && gguf_read_emplace_helper<uint32_t>   (gr, ctx->kv, key, is_array, n); break;590                case GGUF_TYPE_INT32:   ok = ok && gguf_read_emplace_helper<int32_t>    (gr, ctx->kv, key, is_array, n); break;591                case GGUF_TYPE_FLOAT32: ok = ok && gguf_read_emplace_helper<float>      (gr, ctx->kv, key, is_array, n); break;592                case GGUF_TYPE_BOOL:    ok = ok && gguf_read_emplace_helper<bool>       (gr, ctx->kv, key, is_array, n); break;593                case GGUF_TYPE_STRING:  ok = ok && gguf_read_emplace_helper<std::string>(gr, ctx->kv, key, is_array, n); break;594                case GGUF_TYPE_UINT64:  ok = ok && gguf_read_emplace_helper<uint64_t>   (gr, ctx->kv, key, is_array, n); break;595                case GGUF_TYPE_INT64:   ok = ok && gguf_read_emplace_helper<int64_t>    (gr, ctx->kv, key, is_array, n); break;596                case GGUF_TYPE_FLOAT64: ok = ok && gguf_read_emplace_helper<double>     (gr, ctx->kv, key, is_array, n); break;597                case GGUF_TYPE_ARRAY:598                default:599                    {600                        GGML_LOG_ERROR("%s: key '%s' has invalid GGUF type %d\n", __func__, key.c_str(), type);601                        ok = false;602                    } break;603            }604        }605 606        if (!ok) {607            GGML_LOG_ERROR("%s: failed to read key-value pairs\n", __func__);608            gguf_free(ctx);609            return nullptr;610        }611        GGML_ASSERT(int64_t(ctx->kv.size()) == n_kv);612 613        const int alignment_idx = gguf_find_key(ctx, GGUF_KEY_GENERAL_ALIGNMENT);614        ctx->alignment = alignment_idx == -1 ? GGUF_DEFAULT_ALIGNMENT : gguf_get_val_u32(ctx, alignment_idx);615 616        if (ctx->alignment == 0 || (ctx->alignment & (ctx->alignment - 1)) != 0) {617            GGML_LOG_ERROR("%s: alignment %zu is not a power of 2\n", __func__, ctx->alignment);618            gguf_free(ctx);619            return nullptr;620        }621    }622 623    // read the tensor info624    for (int64_t i = 0; ok && i < n_tensors; ++i) {625        struct gguf_tensor_info info;626 627        // tensor name628        {629            std::string name;630            try {631                ok = ok && gr.read(name);632            } catch (std::length_error &) {633                GGML_LOG_ERROR("%s: encountered length_error while reading tensor name %" PRIi64 "\n", __func__, i);634                ok = false;635            } catch (std::bad_alloc &) {636                GGML_LOG_ERROR("%s: encountered bad_alloc error while reading tensor name %" PRIi64 "\n", __func__, i);637                ok = false;638            }639            if (name.length() >= GGML_MAX_NAME) {640                GGML_LOG_ERROR("%s: tensor name %" PRIi64 " is too long: %zu >= %d\n", __func__, i, name.length(), GGML_MAX_NAME);641                ok = false;642                break;643            }644            ggml_set_name(&info.t, name.c_str());645 646            // make sure there are no duplicate tensor names647            for (int64_t j = 0; ok && j < i; ++j) {648                if (strcmp(info.t.name, ctx->info[j].t.name) == 0) {649                    GGML_LOG_ERROR("%s: duplicate tensor name '%s' for tensors %" PRIi64 " and %" PRIi64 "\n", __func__, info.t.name, j, i);650                    ok = false;651                    break;652                }653            }654        }655        if (!ok) {656            break;657        }658 659        // tensor shape660        {661            uint32_t n_dims = 0;662            ok = ok && gr.read(n_dims);663            if (n_dims > GGML_MAX_DIMS) {664                GGML_LOG_ERROR("%s: tensor '%s' has invalid number of dimensions: %" PRIu32 " > %" PRIu32 "\n",665                    __func__, info.t.name, n_dims, GGML_MAX_DIMS);666                ok = false;667                break;668            }669            for (uint32_t j = 0; ok && j < GGML_MAX_DIMS; ++j) {670                info.t.ne[j] = 1;671                if (j < n_dims) {672                    ok = ok && gr.read(info.t.ne[j]);673                }674 675                // check that all ne are non-negative676                if (info.t.ne[j] < 0) {677                    GGML_LOG_ERROR("%s: tensor '%s' dimension %" PRIu32 " has invalid number of elements: %" PRIi64 " < 0\n",678                        __func__, info.t.name, j, info.t.ne[j]);679                    ok = false;680                    break;681                }682            }683 684            // check that the total number of elements is representable685            if (ok && ((INT64_MAX/info.t.ne[1] <= info.t.ne[0]) ||686                       (INT64_MAX/info.t.ne[2] <= info.t.ne[0]*info.t.ne[1]) ||687                       (INT64_MAX/info.t.ne[3] <= info.t.ne[0]*info.t.ne[1]*info.t.ne[2]))) {688 689                GGML_LOG_ERROR("%s: total number of elements in tensor '%s' with shape "690                    "(%" PRIi64 ", %" PRIi64 ", %" PRIi64 ", %" PRIi64 ") is >= %" PRIi64 "\n",691                    __func__, info.t.name, info.t.ne[0], info.t.ne[1], info.t.ne[2], info.t.ne[3], INT64_MAX);692                ok = false;693                break;694            }695        }696        if (!ok) {697            break;698        }699 700        // tensor type701        {702            ok = ok && gr.read(info.t.type);703 704            // check that tensor type is within defined range705            if (info.t.type < 0 || info.t.type >= GGML_TYPE_COUNT) {706                GGML_LOG_ERROR("%s: tensor '%s' has invalid ggml type %d. should be in [0, %d)\n",707                    __func__, info.t.name, info.t.type, GGML_TYPE_COUNT);708                ok = false;709                break;710            }711            const size_t  type_size = ggml_type_size(info.t.type);712            const int64_t blck_size = ggml_blck_size(info.t.type);713 714            // check that row size is divisible by block size715            if (blck_size == 0 || info.t.ne[0] % blck_size != 0) {716                GGML_LOG_ERROR("%s: tensor '%s' of type %d (%s) has %" PRId64 " elements per row, "717                    "not a multiple of block size (%" PRId64 ")\n",718                    __func__, info.t.name, (int) info.t.type, ggml_type_name(info.t.type), info.t.ne[0], blck_size);719                ok = false;720                break;721            }722 723            // check that the size of the tensor in bytes is representable724            if (ok && uint64_t(ggml_nelements(&info.t)/ggml_blck_size(info.t.type)) > SIZE_MAX/ggml_type_size(info.t.type)) {725                GGML_LOG_ERROR("%s: tensor '%s' with shape (%" PRIi64 ", %" PRIi64 ", %" PRIi64 ", %" PRIi64 ") has a size in bytes > %zu\n",726                    __func__, info.t.name, info.t.ne[0], info.t.ne[1], info.t.ne[2], info.t.ne[3], SIZE_MAX);727                ok = false;728                break;729            }730 731            // calculate byte offsets given the tensor shape and type732            info.t.nb[0] = type_size;733            info.t.nb[1] = info.t.nb[0]*(info.t.ne[0]/blck_size);734            for (int j = 2; j < GGML_MAX_DIMS; ++j) {735                info.t.nb[j] = info.t.nb[j - 1]*info.t.ne[j - 1];736            }737        }738        if (!ok) {739            break;740        }741 742        // tensor data offset within buffer743        ok = ok && gr.read(info.offset);744 745        ctx->info.push_back(info);746    }747 748    if (!ok) {749        GGML_LOG_ERROR("%s: failed to read tensor info\n", __func__);750        gguf_free(ctx);751        return nullptr;752    }753    GGML_ASSERT(int64_t(ctx->info.size()) == n_tensors);754 755    // we require the data section to be aligned, so take into account any padding756    if (n_tensors > 0 && !gr.seek(GGML_PAD(gr.tell(), ctx->alignment))) {757        GGML_LOG_ERROR("%s: failed to seek to beginning of data section\n", __func__);758        gguf_free(ctx);759        return nullptr;760    }761 762    // store the current file offset - this is where the data section starts763    ctx->offset = gr.tell();764 765    // compute the total size of the data section, taking into account the alignment766    {767        ctx->size = 0;768        for (size_t i = 0; i < ctx->info.size(); ++i) {769            const gguf_tensor_info & ti = ctx->info[i];770            if (ti.offset != ctx->size) {771                GGML_LOG_ERROR("%s: tensor '%s' has offset %" PRIu64 ", expected %zu\n",772                    __func__, ti.t.name, ti.offset, ctx->size);773                GGML_LOG_ERROR("%s: failed to read tensor data\n", __func__);774                gguf_free(ctx);775                return nullptr;776            }777            size_t padded_size = GGML_PAD(ggml_nbytes(&ti.t), ctx->alignment);778            if (SIZE_MAX - ctx->size < padded_size) {779                GGML_LOG_ERROR("%s: tensor '%s' size overflow, cannot accumulate size %zu + %zu\n",780                    __func__, ti.t.name, ctx->size, padded_size);781                gguf_free(ctx);782                return nullptr;783            }784            ctx->size += padded_size;785        }786    }787 788    // load the tensor data only if requested789    if (params.ctx != nullptr) {790        // if the provided gguf_context is no_alloc, then we create "empty" tensors and do not read the binary blob791        // otherwise, we load the binary blob into the created ggml_context as well, and point the "data" members of792        //   the ggml_tensor structs to the appropriate locations in the binary blob793 794        // compute the exact size needed for the new ggml_context795        size_t mem_size = 0;796        if (params.no_alloc) {797            if (n_tensors != 0 && SIZE_MAX / n_tensors < ggml_tensor_overhead()) {798                GGML_LOG_ERROR("%s: memory size overflow while allocating ggml context\n", __func__);799                gguf_free(ctx);800                return nullptr;801            }802 803            const size_t overhead = n_tensors * ggml_tensor_overhead();804 805            mem_size = overhead;806        } else {807            if ((n_tensors + 1) != 0 && SIZE_MAX / (n_tensors + 1) < ggml_tensor_overhead()) {808                GGML_LOG_ERROR("%s: memory size overflow while allocating ggml context\n", __func__);809                gguf_free(ctx);810                return nullptr;811            }812 813            const size_t overhead = (n_tensors + 1) * ggml_tensor_overhead();814 815            if (SIZE_MAX - overhead < ctx->size) {816                GGML_LOG_ERROR("%s: memory size overflow while allocating ggml context\n", __func__);817                gguf_free(ctx);818                return nullptr;819            }820 821            mem_size = overhead + ctx->size;822        }823 824        struct ggml_init_params pdata = {825            /*mem_size   =*/ mem_size,826            /*mem_buffer =*/ nullptr,827            /*no_alloc   =*/ params.no_alloc,828        };829 830        *params.ctx = ggml_init(pdata);831        if (*params.ctx == nullptr) {832            GGML_LOG_ERROR("%s: failed to initialize ggml context for storing tensors\n", __func__);833            gguf_free(ctx);834            return nullptr;835        }836 837        struct ggml_context * ctx_data = *params.ctx;838 839        struct ggml_tensor * data = nullptr;840 841        if (!params.no_alloc) {842            data = ggml_new_tensor_1d(ctx_data, GGML_TYPE_I8, ctx->size);843 844            ok = ok && data != nullptr;845 846            if (ok) {847                ggml_set_name(data, "GGUF tensor data binary blob");848            }849 850            // read the binary blob with the tensor data851            ok = ok && gr.read(data->data, ctx->size);852 853            if (!ok) {854                GGML_LOG_ERROR("%s: failed to read tensor data binary blob\n", __func__);855                ggml_free(ctx_data);856                *params.ctx = nullptr;857                gguf_free(ctx);858                return nullptr;859            }860 861            ctx->data = data->data;862        }863 864        ggml_set_no_alloc(ctx_data, true);865 866        // create the tensors867        for (size_t i = 0; i < ctx->info.size(); ++i) {868            const struct gguf_tensor_info & info = ctx->info[i];869 870            struct ggml_tensor * cur = ggml_new_tensor(ctx_data, info.t.type, GGML_MAX_DIMS, info.t.ne);871 872            ok = ok && cur != nullptr;873 874            if (!ok) {875                break;876            }877 878            ggml_set_name(cur, info.t.name);879 880            // point the data member to the appropriate location in the binary blob using the tensor info881            if (!params.no_alloc) {882                cur->data = (char *) data->data + info.offset;883            }884        }885 886        if (!ok) {887            GGML_LOG_ERROR("%s: failed to create tensors\n", __func__);888            ggml_free(ctx_data);889            *params.ctx = nullptr;890            gguf_free(ctx);891            return nullptr;892        }893 894        ggml_set_no_alloc(ctx_data, params.no_alloc);895    }896 897    return ctx;898}899 900struct gguf_context * gguf_init_from_callback(gguf_reader_callback_t callback, void * userdata, size_t max_chunk_read, uint64_t max_expected_size, struct gguf_init_params params) {901    if (callback == nullptr) {902        return nullptr;903    }904 905    const struct gguf_reader gr(callback, userdata, max_chunk_read == 0 ? SIZE_MAX : max_chunk_read, 0, max_expected_size);906    return gguf_init_from_reader(gr, params);907}908 909struct gguf_file_reader {910    FILE * file;911    uint64_t offset;912};913 914static size_t gguf_file_reader_callback(void * userdata, void * output, uint64_t offset, size_t len) {915    GGML_ASSERT(len > 0);916 917    gguf_file_reader & reader = *static_cast<gguf_file_reader *>(userdata);918 919    if (reader.offset != offset) {920        if (offset > INT64_MAX || gguf_fseek(reader.file, static_cast<int64_t>(offset), SEEK_SET) != 0) {921            return 0;922        }923 924        reader.offset = offset;925    }926 927    const size_t nread = fread(static_cast<uint8_t *>(output), 1, len, reader.file);928    reader.offset += nread;929    return nread;930}931 932struct gguf_context * gguf_init_from_file_ptr(FILE * file, struct gguf_init_params params) {933    if (!file) {934        return nullptr;935    }936 937    const int64_t cur = gguf_ftell(file);938    if (cur < 0) {939        return nullptr;940    }941 942    gguf_file_reader reader = {943        /*.file   = */ file,944        /*.offset = */ static_cast<uint64_t>(cur),945    };946    const struct gguf_reader gr(gguf_file_reader_callback, &reader, SIZE_MAX, reader.offset, gguf_reader::file_remain(file));947    return gguf_init_from_reader(gr, params);948}949 950struct gguf_buffer_reader {951    const uint8_t * data;952    size_t          size;953};954 955static size_t gguf_buffer_reader_callback(void * userdata, void * output, uint64_t offset, size_t len) {956    GGML_ASSERT(len > 0);957 958    const gguf_buffer_reader & reader = *static_cast<gguf_buffer_reader *>(userdata);959 960    if (offset > reader.size || len > reader.size - offset) {961        return 0;962    }963 964    const size_t data_offset = static_cast<size_t>(offset);965    const size_t nread = std::min(len, reader.size - data_offset);966    memcpy(static_cast<uint8_t *>(output), reader.data + data_offset, nread);967    return nread;968}969 970struct gguf_context * gguf_init_from_buffer(const void * data, size_t size, struct gguf_init_params params) {971    if (data == nullptr || size == 0) {972        return nullptr;973    }974 975    gguf_buffer_reader reader = {976        /*.data = */ static_cast<const uint8_t *>(data),977        /*.size = */ size,978    };979    const struct gguf_reader gr(gguf_buffer_reader_callback, &reader, SIZE_MAX, 0, size);980    return gguf_init_from_reader(gr, params);981}982 983struct gguf_context * gguf_init_from_file(const char * fname, struct gguf_init_params params) {984    FILE * file = ggml_fopen(fname, "rb");985 986    if (!file) {987        GGML_LOG_ERROR("%s: failed to open GGUF file '%s' (%s)\n", __func__, fname, strerror(errno));988        return nullptr;989    }990 991    struct gguf_context * result = gguf_init_from_file_ptr(file, params);992    fclose(file);993    return result;994}995 996void gguf_free(struct gguf_context * ctx) {997    if (ctx == nullptr) {998        return;999    }1000    delete ctx;1001}1002 1003const char * gguf_type_name(enum gguf_type type) {1004    auto it = GGUF_TYPE_NAME.find(type);1005    return it == GGUF_TYPE_NAME.end() ? nullptr : it->second;1006}1007 1008uint32_t gguf_get_version(const struct gguf_context * ctx) {1009    return ctx->version;1010}1011 1012size_t gguf_get_alignment(const struct gguf_context * ctx) {1013    return ctx->alignment;1014}1015 1016size_t gguf_get_data_offset(const struct gguf_context * ctx) {1017    return ctx->offset;1018}1019 1020int64_t gguf_get_n_kv(const struct gguf_context * ctx) {1021    return ctx->kv.size();1022}1023 1024int64_t gguf_find_key(const struct gguf_context * ctx, const char * key) {1025    // return -1 if key not found1026    int64_t keyfound = -1;1027 1028    const int64_t n_kv = gguf_get_n_kv(ctx);1029 1030    for (int64_t i = 0; i < n_kv; ++i) {1031        if (strcmp(key, gguf_get_key(ctx, i)) == 0) {1032            keyfound = i;1033            break;1034        }1035    }1036 1037    return keyfound;1038}1039 1040const char * gguf_get_key(const struct gguf_context * ctx, int64_t key_id) {1041    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1042    return ctx->kv[key_id].get_key().c_str();1043}1044 1045enum gguf_type gguf_get_kv_type(const struct gguf_context * ctx, int64_t key_id) {1046    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1047    return ctx->kv[key_id].is_array ? GGUF_TYPE_ARRAY : ctx->kv[key_id].get_type();1048}1049 1050enum gguf_type gguf_get_arr_type(const struct gguf_context * ctx, int64_t key_id) {1051    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1052    GGML_ASSERT(ctx->kv[key_id].is_array);1053    return ctx->kv[key_id].get_type();1054}1055 1056const void * gguf_get_arr_data(const struct gguf_context * ctx, int64_t key_id) {1057    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1058    GGML_ASSERT(ctx->kv[key_id].get_type() != GGUF_TYPE_STRING);1059    return ctx->kv[key_id].data.data();1060}1061 1062const char * gguf_get_arr_str(const struct gguf_context * ctx, int64_t key_id, size_t i) {1063    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1064    GGML_ASSERT(ctx->kv[key_id].get_type() == GGUF_TYPE_STRING);1065    return ctx->kv[key_id].data_string[i].c_str();1066}1067 1068size_t gguf_get_arr_n(const struct gguf_context * ctx, int64_t key_id) {1069    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1070 1071    if (ctx->kv[key_id].type == GGUF_TYPE_STRING) {1072        return ctx->kv[key_id].data_string.size();1073    }1074 1075    const size_t type_size = gguf_type_size(ctx->kv[key_id].type);1076    GGML_ASSERT(ctx->kv[key_id].data.size() % type_size == 0);1077    return ctx->kv[key_id].data.size() / type_size;1078}1079 1080uint8_t gguf_get_val_u8(const struct gguf_context * ctx, int64_t key_id) {1081    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1082    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1083    return ctx->kv[key_id].get_val<uint8_t>();1084}1085 1086int8_t gguf_get_val_i8(const struct gguf_context * ctx, int64_t key_id) {1087    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1088    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1089    return ctx->kv[key_id].get_val<int8_t>();1090}1091 1092uint16_t gguf_get_val_u16(const struct gguf_context * ctx, int64_t key_id) {1093    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1094    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1095    return ctx->kv[key_id].get_val<uint16_t>();1096}1097 1098int16_t gguf_get_val_i16(const struct gguf_context * ctx, int64_t key_id) {1099    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1100    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1101    return ctx->kv[key_id].get_val<int16_t>();1102}1103 1104uint32_t gguf_get_val_u32(const struct gguf_context * ctx, int64_t key_id) {1105    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1106    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1107    return ctx->kv[key_id].get_val<uint32_t>();1108}1109 1110int32_t gguf_get_val_i32(const struct gguf_context * ctx, int64_t key_id) {1111    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1112    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1113    return ctx->kv[key_id].get_val<int32_t>();1114}1115 1116float gguf_get_val_f32(const struct gguf_context * ctx, int64_t key_id) {1117    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1118    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1119    return ctx->kv[key_id].get_val<float>();1120}1121 1122uint64_t gguf_get_val_u64(const struct gguf_context * ctx, int64_t key_id) {1123    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1124    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1125    return ctx->kv[key_id].get_val<uint64_t>();1126}1127 1128int64_t gguf_get_val_i64(const struct gguf_context * ctx, int64_t key_id) {1129    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1130    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1131    return ctx->kv[key_id].get_val<int64_t>();1132}1133 1134double gguf_get_val_f64(const struct gguf_context * ctx, int64_t key_id) {1135    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1136    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1137    return ctx->kv[key_id].get_val<double>();1138}1139 1140bool gguf_get_val_bool(const struct gguf_context * ctx, int64_t key_id) {1141    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1142    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1143    return ctx->kv[key_id].get_val<bool>();1144}1145 1146const char * gguf_get_val_str(const struct gguf_context * ctx, int64_t key_id) {1147    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1148    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1149    return ctx->kv[key_id].get_val<std::string>().c_str();1150}1151 1152const void * gguf_get_val_data(const struct gguf_context * ctx, int64_t key_id) {1153    GGML_ASSERT(key_id >= 0 && key_id < gguf_get_n_kv(ctx));1154    GGML_ASSERT(ctx->kv[key_id].get_ne() == 1);1155    GGML_ASSERT(ctx->kv[key_id].get_type() != GGUF_TYPE_STRING);1156    return ctx->kv[key_id].data.data();1157}1158 1159int64_t gguf_get_n_tensors(const struct gguf_context * ctx) {1160    return ctx->info.size();1161}1162 1163int64_t gguf_find_tensor(const struct gguf_context * ctx, const char * name) {1164    // return -1 if tensor not found1165    int64_t tensor_id = -1;1166 1167    const int64_t n_tensors = gguf_get_n_tensors(ctx);1168 1169    for (int64_t i = 0; i < n_tensors; ++i) {1170        if (strcmp(name, gguf_get_tensor_name(ctx, i)) == 0) {1171            tensor_id = i;1172            break;1173        }1174    }1175 1176    return tensor_id;1177}1178 1179size_t gguf_get_tensor_offset(const struct gguf_context * ctx, int64_t tensor_id) {1180    GGML_ASSERT(tensor_id >= 0 && tensor_id < gguf_get_n_tensors(ctx));1181    return ctx->info[tensor_id].offset;1182}1183 1184const char * gguf_get_tensor_name(const struct gguf_context * ctx, int64_t tensor_id) {1185    GGML_ASSERT(tensor_id >= 0 && tensor_id < gguf_get_n_tensors(ctx));1186    return ctx->info[tensor_id].t.name;1187}1188 1189const int64_t * gguf_get_tensor_ne(const struct gguf_context * ctx, int64_t tensor_id) {1190    GGML_ASSERT(tensor_id >= 0 && tensor_id < gguf_get_n_tensors(ctx));1191    return ctx->info[tensor_id].t.ne;1192}1193 1194enum ggml_type gguf_get_tensor_type(const struct gguf_context * ctx, int64_t tensor_id) {1195    GGML_ASSERT(tensor_id >= 0 && tensor_id < gguf_get_n_tensors(ctx));1196    return ctx->info[tensor_id].t.type;1197}1198 1199size_t gguf_get_tensor_size(const struct gguf_context * ctx, int64_t tensor_id) {1200    GGML_ASSERT(tensor_id >= 0 && tensor_id < gguf_get_n_tensors(ctx));

Showing the first 1,200 of 1698 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai