Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
results.cpp184 linesDownload Raw Back to results
1#include "ggml-cpp.h"2#include "ggml.h"3#include "gguf.h"4#include "llama.h"5#include "common.h"6#include "arg.h"7#include "log.h"8 9#include <cstdint>10#include <string>11#include <vector>12 13// normalized mean squared error = mse(a, b) / mse(a, 0)14static double nmse(const std::vector<float> & a, const std::vector<float> & b) {15    GGML_ASSERT(a.size() == b.size());16    double mse_a_b = 0.0;17    double mse_a_0 = 0.0;18 19    for (size_t i = 0; i < a.size(); i++) {20        float a_i = a[i];21        float b_i = b[i];22 23        mse_a_b += (a_i - b_i) * (a_i - b_i);24        mse_a_0 += a_i * a_i;25    }26 27    return mse_a_b / mse_a_0;28}29 30static std::vector<float> get_logits(31        llama_model * model, llama_context * lctx, const std::vector<llama_token> & tokens) {32    const uint32_t n_vocab  = llama_vocab_n_tokens(llama_model_get_vocab(model));33    const uint32_t n_ctx    = llama_n_ctx(lctx);34    const uint32_t n_tokens = tokens.size();35    llama_batch batch = llama_batch_init(n_ctx, 0, 1);36    GGML_ASSERT(n_tokens <= n_ctx);37    for (uint32_t pos = 0; pos < n_tokens; pos++) {38        common_batch_add(batch, tokens[pos], pos, {0}, true);39    }40    batch.n_tokens = n_tokens;41    if (llama_decode(lctx, batch)) {42        llama_batch_free(batch);43        throw std::runtime_error("failed to decode batch");44    }45 46    std::vector<float> ret;47    ret.reserve(n_tokens*n_vocab);48    for (uint32_t i = 0; i < n_tokens; i++) {49        const float * logits_ith = llama_get_logits_ith(lctx, i);50        for (uint32_t j = 0; j < n_vocab; j++) {51            ret.push_back(logits_ith[j]);52        }53    }54    llama_batch_free(batch);55    return ret;56}57 58int main(int argc, char ** argv) {59    common_params params;60    params.escape = false;61 62    common_init();63 64    if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_RESULTS)) {65        return 1;66    }67    if (params.out_file.empty()) {68        LOG_ERR("%s: an output file must be specified", __func__);69        return 1;70    }71    llama_backend_init();72    llama_numa_init(params.numa);73    common_init_result_ptr llama_init = common_init_from_params(params);74    struct llama_model   * model = llama_init->model();75    struct llama_context * lctx  = llama_init->context();76    if (model == nullptr) {77        LOG_ERR("%s: unable to load model\n", __func__);78        return 1;79    }80    const uint32_t n_vocab = llama_vocab_n_tokens(llama_model_get_vocab(model));81 82    const std::vector<llama_token> tokens_calc = common_tokenize(lctx, params.prompt, true);83    const std::vector<float> logits_calc = get_logits(model, lctx, tokens_calc);84    GGML_ASSERT(logits_calc.size() == tokens_calc.size()*n_vocab);85 86    struct gguf_init_params gguf_params = {87        /*.no_alloc   =*/ true,88        /*.ctx        =*/ nullptr,89    };90    gguf_context_ptr gguf_ctx_model(gguf_init_from_file(params.model.path.c_str(), gguf_params));91 92    if (params.check) {93        LOG_INF("%s: loading results from %s...\n", __func__, params.out_file.c_str());94        gguf_context_ptr gguf_ctx;95        {96            struct gguf_init_params gguf_params = {97                /*no_alloc =*/ true,98                /*ctx      =*/ nullptr,99            };100            gguf_ctx.reset(gguf_init_from_file(params.out_file.c_str(), gguf_params));101        }102        const std::string path_model_disk = gguf_get_val_str(gguf_ctx.get(), gguf_find_key(gguf_ctx.get(), "path_model"));103        GGML_ASSERT(path_model_disk == params.model.path); // TODO better checks104 105        auto load_tensor_data = [&](const std::string & name, void * dst, const size_t size){106            const int64_t tid    = gguf_find_tensor(gguf_ctx.get(), name.c_str());107            const size_t  offset = gguf_get_data_offset(gguf_ctx.get()) + gguf_get_tensor_offset(gguf_ctx.get(), tid);108            GGML_ASSERT(size == gguf_get_tensor_size(gguf_ctx.get(), tid));109 110            FILE * file = ggml_fopen(params.out_file.c_str(), "rb");111            if (file == nullptr) {112                throw std::runtime_error("failed to open results file");113            }114            if (fseek(file, offset, SEEK_SET) != 0) {115                throw std::runtime_error("fseek failed");116            }117            const size_t nbytes_read = fread(dst, 1, size, file);118            if (nbytes_read != size) {119                throw std::runtime_error("fread failed");120            }121        };122 123        std::vector<llama_token> tokens_disk(tokens_calc.size());124        load_tensor_data("tokens", tokens_disk.data(), tokens_disk.size()*sizeof(llama_token));125        GGML_ASSERT(tokens_disk.size() == tokens_calc.size());126        for (size_t i = 0; i < tokens_calc.size(); i++) {127            GGML_ASSERT(tokens_disk[i] == tokens_calc[i]);128        }129 130        std::vector<float> logits_disk(logits_calc.size());131        load_tensor_data("logits", logits_disk.data(), logits_disk.size()*sizeof(float));132        const double nmse_val = nmse(logits_disk, logits_calc);133        LOG_INF("%s: NMSE=%.3e\n", __func__, nmse_val);134 135        if (nmse_val > 1e-6) {136            printf("\033[1;31mFAIL\033[0m\n");137            return 1;138        }139 140        printf("\033[1;32mOK\033[0m\n");141        return 0;142    }143 144    ggml_context_ptr ggml_ctx_calc;145    {146        const size_t size_tokens = tokens_calc.size()*sizeof(llama_token) + ggml_tensor_overhead();147        const size_t size_logits = logits_calc.size()*sizeof(float)  + ggml_tensor_overhead();148        struct ggml_init_params params = {149            /*.mem_size   =*/ size_tokens + size_logits,150            /*.mem_buffer =*/ nullptr,151            /*.no_alloc   =*/ false,152        };153        ggml_ctx_calc.reset(ggml_init(params));154    }155 156    gguf_context_ptr gguf_ctx(gguf_init_empty());157    gguf_set_val_str(gguf_ctx.get(), "path_model", params.model.path.c_str());158    {159        ggml_tensor * t_tokens = ggml_new_tensor_1d(ggml_ctx_calc.get(), GGML_TYPE_I32, tokens_calc.size());160        ggml_set_name(t_tokens, "tokens");161        int32_t * tokens_data = (int32_t *) t_tokens->data;162        for (uint32_t i = 0; i < tokens_calc.size(); i++) {163            tokens_data[i] = tokens_calc[i];164        }165        gguf_add_tensor(gguf_ctx.get(), t_tokens);166    }167    {168        ggml_tensor * t_logits = ggml_new_tensor_2d(ggml_ctx_calc.get(), GGML_TYPE_F32, tokens_calc.size(), n_vocab);169        ggml_set_name(t_logits, "logits");170        float * logits_data = ggml_get_data_f32(t_logits);171        for (uint32_t i = 0; i < tokens_calc.size(); i++) {172            const float * logits_ith = llama_get_logits_ith(lctx, i);173            for (uint32_t j = 0; j < n_vocab; j++) {174                logits_data[i*n_vocab + j] = logits_ith[j];175            }176        }177        gguf_add_tensor(gguf_ctx.get(), t_logits);178    }179    LOG_INF("%s: writing results to %s...\n", __func__, params.out_file.c_str());180    gguf_write_to_file(gguf_ctx.get(), params.out_file.c_str(), /*only_meta =*/ false);181    return 0;182}183 184 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai