Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
test-backend-ops.cpp10640 linesDownload Raw Back to tests
1// This file defines tests for various GGML ops and backends.2// For the forward pass it asserts that the results of multiple backends computing the same GGML ops are consistent.3// For the backward pass it asserts that the gradients from backpropagation are consistent4// with the gradients obtained via the method of finite differences ("grad" mode, this is optional).5// It is also possible to check the performance ("perf" mode).6//7// this file has three sections: Section 1 does general setup, section 2 defines the GGML ops to be tested,8// and section 3 defines which tests to run.9// Quick start for adding a new GGML op: Go to section 2 and create a struct that inherits from test_case,10// then go to section 3 and add an instantiation of your struct.11 12 13// ##############################14// ## Section 1: General Setup ##15// ##############################16 17 18#include "ggml.h"19#include "ggml-alloc.h"20#include "ggml-backend.h"21#include "ggml-cpp.h"22 23#include <algorithm>24#include <atomic>25#include <array>26#include <cfloat>27#include <cinttypes>28#include <cstdarg>29#include <cstdint>30#include <cstdio>31#include <cstdlib>32#include <cstring>33#include <ctime>34#include <future>35#include <fstream>36#include <memory>37#include <mutex>38#include <random>39#include <regex>40#include <set>41#include <sstream>42#include <string>43#include <string_view>44#include <thread>45#include <vector>46#include <unordered_map>47 48#ifdef __EMSCRIPTEN__49#   define N_THREADS 150#else51#   define N_THREADS std::thread::hardware_concurrency()52#endif53 54static void init_tensor_uniform(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) {55    size_t nels = ggml_nelements(tensor);56    std::vector<float> data(nels);57    {58        // parallel initialization59        static const size_t n_threads = N_THREADS;60 61        auto init_thread = [&](size_t start, size_t end) {62            thread_local std::default_random_engine gen(std::random_device{}());63            std::uniform_real_distribution<float> distribution(min, max);64            for (size_t i = start; i < end; i++) {65                data[i] = distribution(gen);66            }67        };68 69        if (n_threads == 1) {70            init_thread(0, nels);71        } else {72            std::vector<std::future<void>> tasks;73            tasks.reserve(n_threads);74            for (size_t i = 0; i < n_threads; i++) {75                size_t start =     i*nels/n_threads;76                size_t end   = (i+1)*nels/n_threads;77                tasks.push_back(std::async(std::launch::async, init_thread, start, end));78            }79            for (auto & t : tasks) {80                t.get();81            }82        }83    }84 85    if (tensor->type == GGML_TYPE_F32 || tensor->type == GGML_TYPE_I32) {86        ggml_backend_tensor_set(tensor, data.data(), 0, nels * sizeof(float));87    } else if (ggml_is_quantized(tensor->type) || tensor->type == GGML_TYPE_F16 || tensor->type == GGML_TYPE_BF16) {88        GGML_ASSERT(nels % ggml_blck_size(tensor->type) == 0);89 90         // dummy importance matrix91        std::vector<float> imatrix(tensor->ne[0], 1.0f);92        const float * im = imatrix.data();93        if (!ggml_quantize_requires_imatrix(tensor->type)) {94            // when the imatrix is optional, we want to test both quantization with and without imatrix95            // use one of the random numbers to decide96            if (data[0] > 0.5f*(min + max)) {97                im = nullptr;98            }99        }100 101        std::vector<uint8_t> dataq(ggml_row_size(tensor->type, nels));102        {103            // parallel quantization by block104            size_t blck_size = ggml_blck_size(tensor->type);105            size_t n_blocks = nels / blck_size;106 107            auto quantize_thread = [&](size_t start, size_t end) {108                ggml_quantize_chunk(tensor->type, data.data(), dataq.data(),109                    start * blck_size, end - start, blck_size, im);110            };111 112            const size_t min_blocks_per_thread = 1;113            const size_t n_quant_threads = std::min<size_t>(std::max<size_t>(N_THREADS/2, 1),114                                                            std::max<size_t>(1, n_blocks / min_blocks_per_thread));115 116            if (n_quant_threads == 1) {117                // single-threaded quantization: do all blocks in the current thread118                quantize_thread(0, n_blocks);119            } else {120                std::vector<std::future<void>> tasks;121                tasks.reserve(n_quant_threads);122                for (size_t i = 0; i < n_quant_threads; i++) {123                    size_t start =     i*n_blocks/n_quant_threads;124                    size_t end   = (i+1)*n_blocks/n_quant_threads;125                    tasks.push_back(std::async(std::launch::async, quantize_thread, start, end));126                }127                for (auto & t : tasks) {128                    t.get();129                }130            }131        }132        ggml_backend_tensor_set(tensor, dataq.data(), 0, dataq.size());133    } else if (tensor->type == GGML_TYPE_I8 || tensor->type == GGML_TYPE_I16) {134        // This is going to create some weird integers though.135        ggml_backend_tensor_set(tensor, data.data(), 0, nels * ggml_type_size(tensor->type));136    } else if (tensor->type == GGML_TYPE_I64) {137        // Integers with a size of 8 bytes can be set by mirroring the float data, the specific values are again not really meaningful.138        const size_t nbytes_half = nels * sizeof(float);139        ggml_backend_tensor_set(tensor, data.data(), 0*nbytes_half, nbytes_half);140        ggml_backend_tensor_set(tensor, data.data(), 1*nbytes_half, nbytes_half);141    } else {142        GGML_ABORT("fatal error");143    }144}145 146// generate an F16 mask where certain blocks are randomly masked with -INF value147static void init_tensor_kq_mask(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) {148    GGML_ASSERT(tensor->type == GGML_TYPE_F16);149 150    GGML_TENSOR_LOCALS( int32_t, ne, tensor, ne);151 152    std::vector<float>       data_f32(ne0*ne1*ne2*ne3);153    std::vector<ggml_fp16_t> data_f16(ne0*ne1*ne2*ne3);154 155    std::random_device rd;156    std::mt19937 gen(rd());157    std::uniform_real_distribution<float> dis(min, max);158 159    for (size_t i = 0; i < data_f32.size(); i++) {160        data_f32[i] = dis(gen);161    }162 163    // block size164    const int blck0 = 128;165    const int blck1 = 64;166 167    // number of INF/zero blocks168    const int n_inf_zero_blocks = 0.2*(ne0*ne1*ne2*ne3)/(blck0*blck1);169 170    for (int b = 0; b < n_inf_zero_blocks; b++) {171        const int p3 = (rd() % ne3);172        const int p2 = (rd() % ne2);173        const int p1 = (rd() % ne1);174        const int p0 = (rd() % ne0);175 176        bool inf = rd() & 1;177 178        for (int i1 = 0; i1 < blck1 && p1 + i1 < ne1; i1++) {179            const int idx = p3*ne2*ne1*ne0 + p2*ne1*ne0 + (p1 + i1)*ne0 + p0;180 181            for (int i0 = 0; i0 < blck0 && p0 + i0 < ne0; i0++) {182                data_f32[idx + i0] = inf ? -INFINITY : 0.0f;183            }184        }185    }186 187    ggml_fp32_to_fp16_row(data_f32.data(), data_f16.data(), ne0*ne1*ne2*ne3);188 189    ggml_backend_tensor_set(tensor, data_f16.data(), 0, data_f16.size()*sizeof(ggml_fp16_t));190}191 192// generate a lower triangular matrix193static void init_tensor_tril(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) {194    GGML_ASSERT(tensor->type == GGML_TYPE_F32);195    GGML_ASSERT(tensor->ne[0] == tensor->ne[1]);196 197    GGML_TENSOR_LOCALS(int32_t, ne, tensor, ne);198    GGML_TENSOR_LOCALS(size_t, nb, tensor, nb);199 200    std::vector<float> data_f32(ne0*ne1*ne2*ne3);201 202    std::random_device rd;203    std::mt19937 gen(rd());204    std::uniform_real_distribution<float> dis(min, max);205 206    for (int64_t i3 = 0; i3 < ne3; i3++) {207        for (int64_t i2 = 0; i2 < ne2; i2++) {208            for (int64_t i1 = 0; i1 < ne1; i1++) {209                for (int64_t i0 = 0; i0 < ne0; i0++) {210                    int64_t idx = (i0 * nb0 + i1 * nb1 + i2 * nb2 + i3 * nb3) / sizeof(float);211                    if (i0 <= i1) {212                        data_f32[idx] = dis(gen);213                    } else {214                        data_f32[idx] = 0.0f;215                    }216                }217            }218        }219    }220 221    ggml_backend_tensor_set(tensor, data_f32.data(), 0, ggml_nbytes(tensor));222}223 224static std::vector<float> tensor_to_float(const ggml_tensor * t) {225    std::vector<float> tv;226    tv.reserve(ggml_nelements(t));227 228    std::vector<uint8_t> buf(ggml_nbytes(t));229    ggml_backend_tensor_get(t, buf.data(), 0, ggml_nbytes(t));230 231    const auto * tt = ggml_get_type_traits(t->type);232    size_t bs = ggml_blck_size(t->type);233    std::vector<float> vq(ggml_blck_size(t->type));234    bool quantized = ggml_is_quantized(t->type);235 236    // access elements by index to avoid gaps in views237    for (int64_t i3 = 0; i3 < t->ne[3]; i3++) {238        for (int64_t i2 = 0; i2 < t->ne[2]; i2++) {239            for (int64_t i1 = 0; i1 < t->ne[1]; i1++) {240                for (int64_t i0 = 0; i0 < t->ne[0]; i0 += bs) {241                    size_t i = i3*t->nb[3] + i2*t->nb[2] + i1*t->nb[1] + i0/bs*t->nb[0];242                    if (t->type == GGML_TYPE_F16) {243                        tv.push_back(ggml_fp16_to_fp32(*(ggml_fp16_t*)&buf[i]));244                    } else if (t->type == GGML_TYPE_BF16) {245                        tv.push_back(ggml_bf16_to_fp32(*(ggml_bf16_t*)&buf[i]));246                    } else if (t->type == GGML_TYPE_F32) {247                        tv.push_back(*(float *) &buf[i]);248                    } else if (t->type == GGML_TYPE_I64) {249                        tv.push_back((float)*(int64_t *) &buf[i]);250                    } else if (t->type == GGML_TYPE_I32) {251                        tv.push_back((float)*(int32_t *) &buf[i]);252                    } else if (t->type == GGML_TYPE_I16) {253                        tv.push_back((float)*(int16_t *) &buf[i]);254                    } else if (t->type == GGML_TYPE_I8) {255                        tv.push_back((float)*(int8_t *) &buf[i]);256                    } else if (quantized) {257                        tt->to_float(&buf[i], vq.data(), bs);258                        tv.insert(tv.end(), vq.begin(), vq.end());259                    } else {260                        GGML_ABORT("fatal error");261                    }262                }263            }264        }265    }266 267    return tv;268}269 270// normalized mean squared error = mse(a, b) / mse(a, 0)271static double nmse(const float * a, const float * b, size_t n) {272    double mse_a_b = 0.0;273    double mse_a_0 = 0.0;274 275    for (size_t i = 0; i < n; i++) {276        float a_i = a[i];277        float b_i = b[i];278 279        mse_a_b += (a_i - b_i) * (a_i - b_i);280        mse_a_0 += a_i * a_i;281    }282 283    return mse_a_b / mse_a_0;284}285 286// difference between 2 sets (Jaccard distance, 0 - no difference, 1 - no overlap)287template <typename T>288static double jdst(const T * a, const T * b, size_t n) {289    std::unordered_map<T, size_t> set_a;290    std::unordered_map<T, size_t> set_b;291 292    for (size_t i = 0; i < n; ++i) {293        set_a[a[i]]++;294        set_b[b[i]]++;295    }296 297    size_t diff = 0;298 299    for (const auto & p : set_a) {300        const int64_t na = p.second;301        const int64_t nb = set_b.find(p.first) != set_b.end() ? set_b.at(p.first) : 0;302 303        diff += std::abs(na - nb);304    }305 306    for (const auto & p : set_b) {307        if (set_a.find(p.first) == set_a.end()) {308            diff += p.second;309        }310    }311 312    return (double) diff / (2*n);313}314 315// maximum absolute asymmetry between a and b316// asymmetry: (a - b) / (a + b)317// This is more stable than relative error if one of the values fluctuates towards zero.318// n: number of values to compare.319// expected_vals: optional vector of expected values for a. If expected_vals is not empty, filter out all comparisons where320//     a does not match any of the expected values. Needed for noncontinuous gradients where the numerical calculation can fail.321static double mean_abs_asymm(const float * a, const float * b, const size_t n, const std::vector<float> & expected_vals) {322    double sum = 0.0f;323 324    size_t nvalid = 0;325    for (size_t i = 0; i < n; i++) {326        if (!expected_vals.empty()) {327            bool matches_any = false;328            for (const float & ev : expected_vals) {329                if (fabsf(a[i] - ev) < 1e-3f) {330                    matches_any = true;331                    break;332                }333            }334            if (!matches_any) {335                continue;336            }337        }338 339        const float asymm = (a[i] - b[i]) / (a[i] + b[i]);340 341        sum += fabsf(asymm);342        nvalid++;343    }344 345    return sum/nvalid;346}347 348// utils for printing the variables of the test cases349 350static std::string var_to_str(const std::string & x) {351    return x;352}353 354template<typename T>355static std::string var_to_str(const T & x) {356    return std::to_string(x);357}358 359template<typename T, size_t N>360static std::string var_to_str(const T (&x)[N]) {361    std::string s = "[";362    for (size_t i = 0; i < N; i++) {363        if (i > 0) {364            s += ",";365        }366        s += var_to_str(x[i]);367    }368    s += "]";369    return s;370}371 372template<typename T, size_t N>373static std::string var_to_str(const std::array<T, N> & x) {374    std::string s = "[";375    for (size_t i = 0; i < N; i++) {376        if (i > 0) {377            s += ",";378        }379        s += var_to_str(x[i]);380    }381    s += "]";382    return s;383}384 385static std::string var_to_str(ggml_type type) {386    return ggml_type_name(type);387}388 389static std::string var_to_str(ggml_prec prec) {390    return prec == GGML_PREC_F32 ? "f32" : "def";391}392 393static std::string var_to_str(ggml_op_pool pool) {394    switch (pool) {395        case GGML_OP_POOL_AVG:  return "avg";396        case GGML_OP_POOL_MAX:  return "max";397        default:                return std::to_string(pool);398    }399}400 401static std::string var_to_str(ggml_scale_mode mode) {402    std::string str;403    switch (mode & 0xFF) {404        case GGML_SCALE_MODE_NEAREST:  str = "nearest"; break;405        case GGML_SCALE_MODE_BILINEAR: str = "bilinear"; break;406        case GGML_SCALE_MODE_BICUBIC:  str = "bicubic"; break;407        default:                       str = std::to_string(mode); break;408    }409    if (mode & GGML_SCALE_FLAG_ALIGN_CORNERS) {410        str += "|align_corners";411    }412    if (mode & GGML_SCALE_FLAG_ANTIALIAS) {413        str += "|antialias";414    }415    return str;416}417 418#define VAR_TO_STR(x) (#x "=" + var_to_str(x))419 420#define VARS_TO_STR1(a) VAR_TO_STR(a)421#define VARS_TO_STR2(a, b) VAR_TO_STR(a) + "," + VAR_TO_STR(b)422#define VARS_TO_STR3(a, b, c) VAR_TO_STR(a) + "," + VARS_TO_STR2(b, c)423#define VARS_TO_STR4(a, b, c, d) VAR_TO_STR(a) + "," + VARS_TO_STR3(b, c, d)424#define VARS_TO_STR5(a, b, c, d, e) VAR_TO_STR(a) + "," + VARS_TO_STR4(b, c, d, e)425#define VARS_TO_STR6(a, b, c, d, e, f) VAR_TO_STR(a) + "," + VARS_TO_STR5(b, c, d, e, f)426#define VARS_TO_STR7(a, b, c, d, e, f, g) VAR_TO_STR(a) + "," + VARS_TO_STR6(b, c, d, e, f, g)427#define VARS_TO_STR8(a, b, c, d, e, f, g, h) VAR_TO_STR(a) + "," + VARS_TO_STR7(b, c, d, e, f, g, h)428#define VARS_TO_STR9(a, b, c, d, e, f, g, h, i) VAR_TO_STR(a) + "," + VARS_TO_STR8(b, c, d, e, f, g, h, i)429#define VARS_TO_STR10(a, b, c, d, e, f, g, h, i, j) VAR_TO_STR(a) + "," + VARS_TO_STR9(b, c, d, e, f, g, h, i, j)430#define VARS_TO_STR11(a, b, c, d, e, f, g, h, i, j, k) VAR_TO_STR(a) + "," + VARS_TO_STR10(b, c, d, e, f, g, h, i, j, k)431#define VARS_TO_STR12(a, b, c, d, e, f, g, h, i, j, k, l) VAR_TO_STR(a) + "," + VARS_TO_STR11(b, c, d, e, f, g, h, i, j, k, l)432#define VARS_TO_STR13(a, b, c, d, e, f, g, h, i, j, k, l, m) VAR_TO_STR(a) + "," + VARS_TO_STR12(b, c, d, e, f, g, h, i, j, k, l, m)433#define VARS_TO_STR14(a, b, c, d, e, f, g, h, i, j, k, l, m, n) VAR_TO_STR(a) + "," + VARS_TO_STR13(b, c, d, e, f, g, h, i, j, k, l, m, n)434#define VARS_TO_STR15(a, b, c, d, e, f, g, h, i, j, k, l, m, n, o) VAR_TO_STR(a) + "," + VARS_TO_STR14(b, c, d, e, f, g, h, i, j, k, l, m, n, o)435#define VARS_TO_STR16(a, b, c, d, e, f, g, h, i, j, k, l, m, n, o, p) VAR_TO_STR(a) + "," + VARS_TO_STR15(b, c, d, e, f, g, h, i, j, k, l, m, n, o, p)436 437#ifdef GGML_USE_SYCL438static bool inline _isinf(float f) {439    return (*(uint32_t *)&f & 0x7fffffff) == 0x7f800000;440}441#else442static bool inline _isinf(float f) { return std::isinf(f); }443#endif444 445// accept FLT_MAX as infinity446static bool isinf_or_max(float f) {447    return _isinf(f) || f == FLT_MAX || f == -FLT_MAX;448}449 450static bool ggml_is_view_op(enum ggml_op op) {451    return op == GGML_OP_VIEW || op == GGML_OP_RESHAPE || op == GGML_OP_PERMUTE || op == GGML_OP_TRANSPOSE;452}453 454static bool backend_has_feature(ggml_backend_t backend, const char * feature_name) {455    ggml_backend_dev_t dev = ggml_backend_get_device(backend);456    ggml_backend_reg_t reg = ggml_backend_dev_backend_reg(dev);457 458    auto get_features = (ggml_backend_get_features_t) ggml_backend_reg_get_proc_address(reg, "ggml_backend_get_features");459    if (!get_features) {460        return false;461    }462 463    const ggml_backend_feature * features = get_features(reg);464    if (!features) {465        return false;466    }467 468    for (const ggml_backend_feature * f = features; f->name; ++f) {469        if (strcmp(f->name, feature_name) == 0 && strcmp(f->value, "1") == 0) {470            return true;471        }472    }473    return false;474}475 476enum test_mode {477    MODE_TEST,478    MODE_PERF,479    MODE_GRAD,480    MODE_SUPPORT,481};482 483// Output format support similar to llama-bench484enum output_formats { CONSOLE, SQL, CSV };485 486static const char * output_format_str(output_formats format) {487    switch (format) {488        case CONSOLE:489            return "console";490        case SQL:491            return "sql";492        case CSV:493            return "csv";494        default:495            GGML_ABORT("invalid output format");496    }497}498 499static bool output_format_from_str(const std::string & s, output_formats & format) {500    if (s == "console") {501        format = CONSOLE;502    } else if (s == "sql") {503        format = SQL;504    } else if (s == "csv") {505        format = CSV;506    } else {507        return false;508    }509    return true;510}511 512static std::string test_time_now() {513    time_t t = time(NULL);514    struct tm tm_buf;515#ifdef _WIN32516    if (gmtime_s(&tm_buf, &t) != 0) {517        return "";518    }519#else520    if (gmtime_r(&t, &tm_buf) == nullptr) {521        return "";522    }523#endif524    char buf[32];525    if (std::strftime(buf, sizeof(buf), "%FT%TZ", &tm_buf) == 0) {526        return "";527    }528    return buf;529}530 531// Test result structure for SQL output532struct test_result {533    std::string test_time;534    std::string build_commit;535    std::string backend_name;536    std::string op_name;537    std::string op_params;538    std::string test_mode;539    bool        supported;540    bool        passed;541    std::string error_message;542    double      time_us;543    double      flops;544    double      bandwidth_gb_s;545    size_t      memory_kb;546    int         n_runs;547    std::string device_description;548    std::string backend_reg_name;549 550    test_result() {551        // Initialize with default values552        time_us        = 0.0;553        flops          = 0.0;554        bandwidth_gb_s = 0.0;555        memory_kb      = 0;556        n_runs         = 0;557        supported      = false;558        passed         = false;559 560        test_time = test_time_now();561 562        // Set build info563        build_commit = ggml_commit();564    }565 566    test_result(const std::string & backend_name, const std::string & op_name, const std::string & op_params,567                const std::string & test_mode, bool supported, bool passed, const std::string & error_message = "",568                double time_us = 0.0, double flops = 0.0, double bandwidth_gb_s = 0.0, size_t memory_kb = 0,569                int n_runs = 0, const std::string & device_description = "", const std::string & backend_reg_name = "") :570        backend_name(backend_name),571        op_name(op_name),572        op_params(op_params),573        test_mode(test_mode),574        supported(supported),575        passed(passed),576        error_message(error_message),577        time_us(time_us),578        flops(flops),579        bandwidth_gb_s(bandwidth_gb_s),580        memory_kb(memory_kb),581        n_runs(n_runs),582        device_description(device_description),583        backend_reg_name(backend_reg_name) {584        test_time = test_time_now();585 586        // Set build info587        build_commit = ggml_commit();588    }589 590    static const std::vector<std::string> & get_fields() {591        static const std::vector<std::string> fields = {592            "test_time", "build_commit",  "backend_name", "op_name", "op_params",      "test_mode", "supported",593            "passed",    "error_message", "time_us",      "flops",   "bandwidth_gb_s", "memory_kb", "n_runs",594            "device_description", "backend_reg_name"595        };596        return fields;597    }598 599    enum field_type { STRING, BOOL, INT, FLOAT };600 601    static field_type get_field_type(const std::string & field) {602        if (field == "supported" || field == "passed") {603            return BOOL;604        }605        if (field == "memory_kb" || field == "n_runs") {606            return INT;607        }608        if (field == "time_us" || field == "flops" || field == "bandwidth_gb_s") {609            return FLOAT;610        }611        return STRING;612    }613 614    std::vector<std::string> get_values() const {615        return { test_time,616                 build_commit,617                 backend_name,618                 op_name,619                 op_params,620                 test_mode,621                 std::to_string(supported),622                 std::to_string(passed),623                 error_message,624                 std::to_string(time_us),625                 std::to_string(flops),626                 std::to_string(bandwidth_gb_s),627                 std::to_string(memory_kb),628                 std::to_string(n_runs),629                 device_description,630                 backend_reg_name };631    }632};633 634// Printer classes for different output formats635enum class test_status_t { NOT_SUPPORTED, OK, FAIL, SKIPPED };636 637struct test_operation_info {638    std::string   op_name;639    std::string   op_params;640    std::string   backend_name;641    test_status_t status = test_status_t::OK;642    std::string   failure_reason;643 644    // Additional information fields that were previously in separate structs645    std::string error_component;646    std::string error_details;647 648    // Gradient info649    int64_t     gradient_index = -1;650    std::string gradient_param_name;651    float       gradient_value = 0.0f;652 653    // MAA error info654    double maa_error     = 0.0;655    double maa_threshold = 0.0;656 657    // Flags for different types of information658    bool has_error            = false;659    bool has_gradient_info    = false;660    bool has_maa_error        = false;661    bool is_compare_failure   = false;662    bool is_large_tensor_skip = false;663 664    test_operation_info() = default;665 666    test_operation_info(const std::string & op_name, const std::string & op_params, const std::string & backend_name,667                        test_status_t status = test_status_t::OK, const std::string & failure_reason = "") :668        op_name(op_name),669        op_params(op_params),670        backend_name(backend_name),671        status(status),672        failure_reason(failure_reason) {}673 674    // Set error information675    void set_error(const std::string & component, const std::string & details) {676        has_error       = true;677        error_component = component;678        error_details   = details;679        if (status == test_status_t::OK) {680            status = test_status_t::FAIL;681        }682    }683 684    // Set gradient information685    void set_gradient_info(int64_t index, const std::string & param_name, float value) {686        has_gradient_info   = true;687        gradient_index      = index;688        gradient_param_name = param_name;689        gradient_value      = value;690        if (status == test_status_t::OK) {691            status = test_status_t::FAIL;692        }693    }694 695    // Set MAA error information696    void set_maa_error(double error, double threshold) {697        has_maa_error = true;698        maa_error     = error;699        maa_threshold = threshold;700        if (status == test_status_t::OK) {701            status = test_status_t::FAIL;702        }703    }704 705    // Set compare failure706    void set_compare_failure() {707        is_compare_failure = true;708        if (status == test_status_t::OK) {709            status = test_status_t::FAIL;710        }711    }712 713    // Set large tensor skip714    void set_large_tensor_skip() { is_large_tensor_skip = true; }715};716 717struct test_summary_info {718    size_t tests_passed;719    size_t tests_total;720    bool   is_backend_summary = false;  // true for backend summary, false for test summary721 722    test_summary_info() = default;723 724    test_summary_info(size_t tests_passed, size_t tests_total, bool is_backend_summary = false) :725        tests_passed(tests_passed),726        tests_total(tests_total),727        is_backend_summary(is_backend_summary) {}728};729 730struct testing_start_info {731    size_t device_count;732 733    testing_start_info() = default;734 735    testing_start_info(size_t device_count) : device_count(device_count) {}736};737 738struct backend_init_info {739    size_t      device_index;740    size_t      total_devices;741    std::string device_name;742    bool        skipped = false;743    std::string skip_reason;744    std::string description;745    size_t      memory_total_mb = 0;746    size_t      memory_free_mb  = 0;747    bool        has_memory_info = false;748 749    backend_init_info() = default;750 751    backend_init_info(size_t device_index, size_t total_devices, const std::string & device_name, bool skipped = false,752                      const std::string & skip_reason = "", const std::string & description = "",753                      size_t memory_total_mb = 0, size_t memory_free_mb = 0, bool has_memory_info = false) :754        device_index(device_index),755        total_devices(total_devices),756        device_name(device_name),757        skipped(skipped),758        skip_reason(skip_reason),759        description(description),760        memory_total_mb(memory_total_mb),761        memory_free_mb(memory_free_mb),762        has_memory_info(has_memory_info) {}763};764 765struct backend_status_info {766    std::string   backend_name;767    test_status_t status;768 769    backend_status_info() = default;770 771    backend_status_info(const std::string & backend_name, test_status_t status) :772        backend_name(backend_name),773        status(status) {}774};775 776struct overall_summary_info {777    size_t backends_passed;778    size_t backends_total;779    bool   all_passed;780 781    overall_summary_info() = default;782 783    overall_summary_info(size_t backends_passed, size_t backends_total, bool all_passed) :784        backends_passed(backends_passed),785        backends_total(backends_total),786        all_passed(all_passed) {}787};788 789struct printer {790    virtual ~printer() {}791 792    FILE * fout = stdout;793 794    virtual void print_header() {}795 796    virtual void print_test_result(const test_result & result) = 0;797 798    virtual void print_footer() {}799 800    virtual void print_operation(const test_operation_info & info) { (void) info; }801 802    virtual void print_summary(const test_summary_info & info) { (void) info; }803 804    virtual void print_testing_start(const testing_start_info & info) { (void) info; }805 806    virtual void print_backend_init(const backend_init_info & info) { (void) info; }807 808    virtual void print_backend_status(const backend_status_info & info) { (void) info; }809 810    virtual void print_overall_summary(const overall_summary_info & info) { (void) info; }811 812    virtual void print_failed_tests(const std::vector<std::string> & failed_tests) { (void) failed_tests; }813};814 815struct console_printer : public printer {816    void print_test_result(const test_result & result) override {817        if (result.test_mode == "test") {818            print_test_console(result);819        } else if (result.test_mode == "perf") {820            print_perf_console(result);821        } else if (result.test_mode == "support") {822            print_support_console(result);823        }824    }825 826    void print_operation(const test_operation_info & info) override {827        printf("  %s(%s): ", info.op_name.c_str(), info.op_params.c_str());828        fflush(stdout);829 830        // Handle large tensor skip first831        if (info.is_large_tensor_skip) {832            printf("skipping large tensors for speed \n");833            return;834        }835 836        // Handle not supported status837        if (info.status == test_status_t::NOT_SUPPORTED) {838            if (!info.failure_reason.empty()) {839                printf("not supported [%s]\n", info.failure_reason.c_str());840            } else {841                printf("not supported [%s]\n", info.backend_name.c_str());842            }843            return;844        }845 846        // Handle errors and additional information847        if (info.has_error) {848            if (info.error_component == "allocation") {849                fprintf(stderr, "failed to allocate tensors [%s] ", info.backend_name.c_str());850            } else if (info.error_component == "backend") {851                fprintf(stderr, "  Failed to initialize %s backend\n", info.backend_name.c_str());852            } else {853                fprintf(stderr, "Error in %s: %s\n", info.error_component.c_str(), info.error_details.c_str());854            }855        }856 857        // Handle gradient info858        if (info.has_gradient_info) {859            printf("[%s] nonfinite gradient at index %" PRId64 " (%s=%f) ", info.op_name.c_str(), info.gradient_index,860                   info.gradient_param_name.c_str(), info.gradient_value);861        }862 863        // Handle MAA error864        if (info.has_maa_error) {865            printf("[%s] MAA = %.9f > %.9f ", info.op_name.c_str(), info.maa_error, info.maa_threshold);866        }867 868        // Handle compare failure869        if (info.is_compare_failure) {870            printf("compare failed ");871        }872 873        // Print final status874        if (info.status == test_status_t::OK) {875            printf("\033[1;32mOK\033[0m\n");876        } else {877            printf("\033[1;31mFAIL\033[0m\n");878        }879    }880 881    void print_summary(const test_summary_info & info) override {882        if (info.is_backend_summary) {883            printf("%zu/%zu backends passed\n", info.tests_passed, info.tests_total);884        } else {885            printf("  %zu/%zu tests passed\n", info.tests_passed, info.tests_total);886        }887    }888 889    void print_backend_status(const backend_status_info & info) override {890        printf("  Backend %s: ", info.backend_name.c_str());891        if (info.status == test_status_t::OK) {892            printf("\033[1;32mOK\033[0m\n");893        } else {894            printf("\033[1;31mFAIL\033[0m\n");895        }896    }897 898    void print_testing_start(const testing_start_info & info) override {899        printf("Testing %zu devices\n\n", info.device_count);900    }901 902    void print_backend_init(const backend_init_info & info) override {903        printf("Backend %zu/%zu: %s\n", info.device_index + 1, info.total_devices, info.device_name.c_str());904 905        if (info.skipped) {906            printf("  %s\n", info.skip_reason.c_str());907            return;908        }909 910        if (!info.description.empty()) {911            printf("  Device description: %s\n", info.description.c_str());912        }913 914        if (info.has_memory_info) {915            printf("  Device memory: %zu MB (%zu MB free)\n", info.memory_total_mb, info.memory_free_mb);916        }917 918        printf("\n");919    }920 921    void print_overall_summary(const overall_summary_info & info) override {922        printf("%zu/%zu backends passed\n", info.backends_passed, info.backends_total);923        if (info.all_passed) {924            printf("\033[1;32mOK\033[0m\n");925        } else {926            printf("\033[1;31mFAIL\033[0m\n");927        }928    }929 930    void print_failed_tests(const std::vector<std::string> & failed_tests) override {931        if (failed_tests.empty()) {932            return;933        }934 935        printf("\nFailing tests:\n");936        for (const auto & test_name : failed_tests) {937            printf("  %s\n", test_name.c_str());938        }939    }940 941  private:942    void print_test_console(const test_result & result) {943        printf("  %s(%s): ", result.op_name.c_str(), result.op_params.c_str());944        fflush(stdout);945 946        if (!result.supported) {947            printf("not supported [%s] ", result.backend_name.c_str());948            printf("\n");949            return;950        }951 952        if (result.passed) {953            printf("\033[1;32mOK\033[0m\n");954        } else {955            printf("\033[1;31mFAIL\033[0m\n");956        }957    }958 959    void print_perf_console(const test_result & result) {960        int len = printf("  %s(%s): ", result.op_name.c_str(), result.op_params.c_str());961        fflush(stdout);962 963        if (!result.supported) {964            printf("not supported\n");965            return;966        }967 968        // align while also leaving some margin for variations in parameters969        int align = 8;970        int last  = (len + align - 1) / align * align;971        if (last - len < 5) {972            last += align;973        }974        printf("%*s", last - len, "");975 976        printf("    %8d runs - %8.2f us/run - ", result.n_runs, result.time_us);977 978        if (result.flops > 0) {979            auto format_flops = [](double flops) -> std::string {980                char buf[256];981                if (flops >= 1e12) {982                    snprintf(buf, sizeof(buf), "%6.2f TFLOP", flops / 1e12);983                } else if (flops >= 1e9) {984                    snprintf(buf, sizeof(buf), "%6.2f GFLOP", flops / 1e9);985                } else if (flops >= 1e6) {986                    snprintf(buf, sizeof(buf), "%6.2f MFLOP", flops / 1e6);987                } else {988                    snprintf(buf, sizeof(buf), "%6.2f kFLOP", flops / 1e3);989                }990                return buf;991            };992            uint64_t op_flops_per_run = result.flops * result.time_us / 1e6;993            printf("%s/run - \033[1;34m%sS\033[0m", format_flops(op_flops_per_run).c_str(),994                   format_flops(result.flops).c_str());995        } else {996            printf("%8zu kB/run - \033[1;34m%7.2f GB/s\033[0m", result.memory_kb, result.bandwidth_gb_s);997        }998        printf("\n");999    }1000 1001    void print_support_console(const test_result & result) {1002        printf("  %s(%s): ", result.op_name.c_str(), result.op_params.c_str());1003        fflush(stdout);1004 1005        if (result.supported) {1006            printf("\033[1;32mSUPPORTED\033[0m\n");1007        } else {1008            printf("\033[1;31mNOT SUPPORTED\033[0m\n");1009        }1010    }1011};1012 1013struct sql_printer : public printer {1014    static std::string get_sql_field_type(const std::string & field) {1015        switch (test_result::get_field_type(field)) {1016            case test_result::STRING:1017                return "TEXT";1018            case test_result::BOOL:1019            case test_result::INT:1020                return "INTEGER";1021            case test_result::FLOAT:1022                return "REAL";1023            default:1024                GGML_ABORT("invalid field type");1025        }1026    }1027 1028    void print_header() override {1029        std::vector<std::string> fields = test_result::get_fields();1030        fprintf(fout, "CREATE TABLE IF NOT EXISTS test_backend_ops (\n");1031        for (size_t i = 0; i < fields.size(); i++) {1032            fprintf(fout, "  %s %s%s\n", fields[i].c_str(), get_sql_field_type(fields[i]).c_str(),1033                    i < fields.size() - 1 ? "," : "");1034        }1035        fprintf(fout, ");\n\n");1036    }1037 1038    void print_test_result(const test_result & result) override {1039        fprintf(fout, "INSERT INTO test_backend_ops (");1040        std::vector<std::string> fields = test_result::get_fields();1041        for (size_t i = 0; i < fields.size(); i++) {1042            fprintf(fout, "%s%s", fields[i].c_str(), i < fields.size() - 1 ? ", " : "");1043        }1044        fprintf(fout, ") VALUES (");1045        std::vector<std::string> values = result.get_values();1046        for (size_t i = 0; i < values.size(); i++) {1047            fprintf(fout, "'%s'%s", values[i].c_str(), i < values.size() - 1 ? ", " : "");1048        }1049        fprintf(fout, ");\n");1050    }1051};1052 1053struct csv_printer : public printer {1054    void print_header() override {1055 1056        std::vector<std::string> fields     = test_result::get_fields();1057        std::vector<std::string> fields_csv = get_fields_csv();1058        for (size_t i = 0; i < fields.size(); i++) {1059            if (std::find(std::begin(fields_csv), std::end(fields_csv), fields[i]) == std::end(fields_csv)) {1060                continue;1061            }1062            printf("\"%s\"%s", fields[i].c_str(), i < fields.size() - 1 ? "," : "");1063        }1064        printf("\n");1065    }1066 1067    void print_test_result(const test_result & result) override {1068 1069        std::vector<std::string> values     = result.get_values();1070        std::vector<std::string> fields     = test_result::get_fields();1071        std::vector<std::string> fields_csv = get_fields_csv();1072 1073        for (size_t i = 0; i < values.size(); i++) {1074 1075            if (std::find(std::begin(fields_csv), std::end(fields_csv), fields[i]) == std::end(fields_csv)) {1076                continue;1077            }1078 1079            // Escape quotes and wrap in quotes for CSV1080            std::string escaped_value = values[i];1081            size_t pos = 0;1082            while ((pos = escaped_value.find("\"", pos)) != std::string::npos) {1083                escaped_value.replace(pos, 1, "\"\"");1084                pos += 2;1085            }1086            printf("\"%s\"%s", escaped_value.c_str(), i < values.size() - 1 ? "," : "");1087        }1088        printf("\n");1089    }1090 1091    static std::vector<std::string> get_fields_csv() {1092        return {1093            "op_name",1094            "op_params",1095            "supported",1096            "error_message",1097            "test_mode",1098            "backend_reg_name",1099            "backend_name",1100        };1101    }1102 1103};1104 1105static std::unique_ptr<printer> create_printer(output_formats format) {1106    switch (format) {1107        case CONSOLE:1108            return std::make_unique<console_printer>();1109        case SQL:1110            return std::make_unique<sql_printer>();1111        case CSV:1112            return std::make_unique<csv_printer>();1113    }1114    GGML_ABORT("invalid output format");1115}1116 1117static std::mutex g_test_output_mutex;1118 1119static void print_test_result_locked(printer * output_printer, const test_result & result) {1120    if (output_printer == nullptr) {1121        return;1122    }1123 1124    std::lock_guard<std::mutex> guard(g_test_output_mutex);1125    output_printer->print_test_result(result);1126}1127 1128struct test_case {1129    virtual ~test_case() {}1130 1131    virtual std::string op_desc(ggml_tensor * t) {1132        return ggml_op_desc(t);1133    }1134 1135    virtual std::string vars() {1136        return "";1137    }1138 1139    virtual ggml_tensor * build_graph(ggml_context * ctx) = 0;1140    virtual ggml_tensor * build_graph(ggml_context * ctx, ggml_context * ctx_weights) {1141        GGML_UNUSED(ctx_weights);1142        return build_graph(ctx);1143    }1144 1145    virtual double max_nmse_err() {1146        return 1e-7;1147    }1148 1149    virtual double max_nmse_err(ggml_backend_t backend) {1150        ggml_backend_reg_t reg = ggml_backend_dev_backend_reg(ggml_backend_get_device(backend));1151        // See https://github.com/ggml-org/llama.cpp/pull/22976 for explanation.1152        if (contains_f16 && strcmp(ggml_backend_reg_name(reg), "WebGPU") == 0) {1153            return std::max(max_nmse_err(), 1e-6);1154        }1155        return max_nmse_err();1156    }1157 1158    virtual double max_maa_err() {1159        return 1e-4;1160    }1161 1162    virtual double max_err() {1163        return max_nmse_err();1164    }1165 1166    virtual double max_err(ggml_backend_t backend) {1167        return max_nmse_err(backend);1168    }1169 1170    virtual double err(const float * a, const float * b, size_t n) {1171        return nmse(a, b, n);1172    }1173 1174    virtual float grad_eps() {1175        return 1e-1f;1176    }1177 1178    // If false, estimate gradient with 2 points, neglects 3rd order derivative and higher.1179    // If true,  estimate gradient with 4 points, neglects 5th order derivative and higher.1180    virtual bool grad_precise() {1181        return false;1182    }1183 1184    // Skip gradient checks if total number of gradients to be checked is larger than this (to speed up the tests).1185    virtual int64_t grad_nmax() {1186        return 10000;1187    }1188 1189    // No effect if empty.1190    // If not empty, skip all gradient checks where the numerical result does not match any of the values.1191    // Needed for dealing with noncontinuous gradients (e.g. ReLU) where estimation using finite differences is unreliable.1192    virtual std::vector<float> grad_expect() {1193        return {};1194    }1195 1196    virtual void initialize_tensors(ggml_context * ctx) {1197        for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {1198            init_tensor_uniform(t);1199        }1200    }

Showing the first 1,200 of 10640 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai