Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
test-backend-ops.cpp4517 linesDownload Raw Back to tests
1// This file defines tests for various GGML ops and backends.2// For the forward pass it asserts that the results of multiple backends computing the same GGML ops are consistent.3// For the backward pass it asserts that the gradients from backpropagation are consistent4// with the gradients obtained via the method of finite differences ("grad" mode, this is optional).5// It is also possible to check the performance ("perf" mode).6//7// this file has three sections: Section 1 does general setup, section 2 defines the GGML ops to be tested,8// and section 3 defines which tests to run.9// Quick start for adding a new GGML op: Go to section 2 and create a struct that inherits from test_case,10// then go to section 3 and add an instantiation of your struct.11 12 13// ##############################14// ## Section 1: General Setup ##15// ##############################16 17 18#include <ggml.h>19#include <ggml-alloc.h>20#include <ggml-backend.h>21 22#include <algorithm>23#include <array>24#include <cfloat>25#include <cstdint>26#include <cstring>27#include <cinttypes>28#include <memory>29#include <random>30#include <stdio.h>31#include <stdlib.h>32#include <string>33#include <thread>34#include <future>35#include <vector>36 37static void init_tensor_uniform(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) {38    size_t nels = ggml_nelements(tensor);39    std::vector<float> data(nels);40    {41        // parallel initialization42        static const size_t n_threads = std::thread::hardware_concurrency();43        // static RNG initialization (revisit if n_threads stops being constant)44        static std::vector<std::default_random_engine> generators = []() {45            std::random_device rd;46            std::vector<std::default_random_engine> vec;47            vec.reserve(n_threads);48            //for (size_t i = 0; i < n_threads; i++) { vec.emplace_back(1234 + i); } // fixed seed49            for (size_t i = 0; i < n_threads; i++) { vec.emplace_back(rd()); }50            return vec;51        }();52 53        auto init_thread = [&](size_t ith, size_t start, size_t end) {54            std::uniform_real_distribution<float> distribution(min, max);55            auto & gen = generators[ith];56            for (size_t i = start; i < end; i++) {57                data[i] = distribution(gen);58            }59        };60 61        std::vector<std::future<void>> tasks;62        tasks.reserve(n_threads);63        for (size_t i = 0; i < n_threads; i++) {64            size_t start =     i*nels/n_threads;65            size_t end   = (i+1)*nels/n_threads;66            tasks.push_back(std::async(std::launch::async, init_thread, i, start, end));67        }68        for (auto & t : tasks) {69            t.get();70        }71    }72 73    if (tensor->type == GGML_TYPE_F32 || tensor->type == GGML_TYPE_I32) {74        ggml_backend_tensor_set(tensor, data.data(), 0, nels * sizeof(float));75    } else if (ggml_is_quantized(tensor->type) || tensor->type == GGML_TYPE_F16 || tensor->type == GGML_TYPE_BF16) {76        GGML_ASSERT(nels % ggml_blck_size(tensor->type) == 0);77 78         // dummy importance matrix79        std::vector<float> imatrix(tensor->ne[0], 1.0f);80        const float * im = imatrix.data();81        if (!ggml_quantize_requires_imatrix(tensor->type)) {82            // when the imatrix is optional, we want to test both quantization with and without imatrix83            // use one of the random numbers to decide84            if (data[0] > 0.5f*(min + max)) {85                im = nullptr;86            }87        }88 89        std::vector<uint8_t> dataq(ggml_row_size(tensor->type, nels));90        {91            // parallel quantization by block92            size_t blck_size = ggml_blck_size(tensor->type);93            size_t n_blocks = nels / blck_size;94 95            auto quantize_thread = [&](size_t start, size_t end) {96                ggml_quantize_chunk(tensor->type, data.data(), dataq.data(),97                    start * blck_size, end - start, blck_size, im);98            };99 100            const size_t min_blocks_per_thread = 1;101            const size_t n_threads = std::min<size_t>(std::thread::hardware_concurrency()/2,102                                                      std::max<size_t>(1, n_blocks / min_blocks_per_thread));103            std::vector<std::future<void>> tasks;104            tasks.reserve(n_threads);105            for (size_t i = 0; i < n_threads; i++) {106                size_t start =     i*n_blocks/n_threads;107                size_t end   = (i+1)*n_blocks/n_threads;108                tasks.push_back(std::async(std::launch::async, quantize_thread, start, end));109            }110            for (auto & t : tasks) {111                t.get();112            }113        }114        ggml_backend_tensor_set(tensor, dataq.data(), 0, dataq.size());115    } else if (tensor->type == GGML_TYPE_I8 || tensor->type == GGML_TYPE_I16 || tensor->type == GGML_TYPE_I32) {116        // This is going to create some weird integers though.117        ggml_backend_tensor_set(tensor, data.data(), 0, ggml_nbytes(tensor));118    } else if (tensor->type == GGML_TYPE_I64) {119        // Integers with a size of 8 bytes can be set by mirroring the float data, the specific values are again not really meaningful.120        const size_t nbytes_half = ggml_nbytes(tensor)/2;121        ggml_backend_tensor_set(tensor, data.data(), 0*nbytes_half, nbytes_half);122        ggml_backend_tensor_set(tensor, data.data(), 1*nbytes_half, nbytes_half);123    } else {124        GGML_ABORT("fatal error");125    }126}127 128static std::vector<float> tensor_to_float(const ggml_tensor * t) {129    std::vector<float> tv;130    tv.reserve(ggml_nelements(t));131 132    std::vector<uint8_t> buf(ggml_nbytes(t));133    ggml_backend_tensor_get(t, buf.data(), 0, ggml_nbytes(t));134 135    const auto * tt = ggml_get_type_traits(t->type);136    size_t bs = ggml_blck_size(t->type);137    std::vector<float> vq(ggml_blck_size(t->type));138    bool quantized = ggml_is_quantized(t->type);139 140    // access elements by index to avoid gaps in views141    for (int64_t i3 = 0; i3 < t->ne[3]; i3++) {142        for (int64_t i2 = 0; i2 < t->ne[2]; i2++) {143            for (int64_t i1 = 0; i1 < t->ne[1]; i1++) {144                for (int64_t i0 = 0; i0 < t->ne[0]; i0 += bs) {145                    size_t i = i3*t->nb[3] + i2*t->nb[2] + i1*t->nb[1] + i0/bs*t->nb[0];146                    if (t->type == GGML_TYPE_F16) {147                        tv.push_back(ggml_fp16_to_fp32(*(ggml_fp16_t*)&buf[i]));148                    } else if (t->type == GGML_TYPE_BF16) {149                        tv.push_back(ggml_bf16_to_fp32(*(ggml_bf16_t*)&buf[i]));150                    } else if (t->type == GGML_TYPE_F32) {151                        tv.push_back(*(float *) &buf[i]);152                    } else if (t->type == GGML_TYPE_I64) {153                        tv.push_back((float)*(int64_t *) &buf[i]);154                    } else if (t->type == GGML_TYPE_I32) {155                        tv.push_back((float)*(int32_t *) &buf[i]);156                    } else if (t->type == GGML_TYPE_I16) {157                        tv.push_back((float)*(int16_t *) &buf[i]);158                    } else if (t->type == GGML_TYPE_I8) {159                        tv.push_back((float)*(int8_t *) &buf[i]);160                    } else if (quantized) {161                        tt->to_float(&buf[i], vq.data(), bs);162                        tv.insert(tv.end(), vq.begin(), vq.end());163                    } else {164                        GGML_ABORT("fatal error");165                    }166                }167            }168        }169    }170 171    return tv;172}173 174// normalized mean squared error = mse(a, b) / mse(a, 0)175static double nmse(const float * a, const float * b, size_t n) {176    double mse_a_b = 0.0;177    double mse_a_0 = 0.0;178 179    for (size_t i = 0; i < n; i++) {180        float a_i = a[i];181        float b_i = b[i];182 183        mse_a_b += (a_i - b_i) * (a_i - b_i);184        mse_a_0 += a_i * a_i;185    }186 187    return mse_a_b / mse_a_0;188}189 190// maximum absolute asymmetry between a and b191// asymmetry: (a - b) / (a + b)192// This is more stable than relative error if one of the values fluctuates towards zero.193// n: number of values to compare.194// expected_vals: optional vector of expected values for a. If expected_vals is not empty, filter out all comparisons where195//     a does not match any of the expected values. Needed for noncontinuous gradients where the numerical calculation can fail.196static double mean_abs_asymm(const float * a, const float * b, const size_t n, const std::vector<float> & expected_vals) {197    double sum = 0.0f;198 199    size_t nvalid = 0;200    for (size_t i = 0; i < n; i++) {201        if (!expected_vals.empty()) {202            bool matches_any = false;203            for (const float & ev : expected_vals) {204                if (fabsf(a[i] - ev) < 1e-3f) {205                    matches_any = true;206                    break;207                }208            }209            if (!matches_any) {210                continue;211            }212        }213 214        const float asymm = (a[i] - b[i]) / (a[i] + b[i]);215 216        sum += fabsf(asymm);217        nvalid++;218    }219 220    return sum/nvalid;221}222 223// utils for printing the variables of the test cases224 225template<typename T>226static std::string var_to_str(const T & x) {227    return std::to_string(x);228}229 230template<typename T, size_t N>231static std::string var_to_str(const T (&x)[N]) {232    std::string s = "[";233    for (size_t i = 0; i < N; i++) {234        if (i > 0) {235            s += ",";236        }237        s += var_to_str(x[i]);238    }239    s += "]";240    return s;241}242 243template<typename T, size_t N>244static std::string var_to_str(const std::array<T, N> & x) {245    std::string s = "[";246    for (size_t i = 0; i < N; i++) {247        if (i > 0) {248            s += ",";249        }250        s += var_to_str(x[i]);251    }252    s += "]";253    return s;254}255 256static std::string var_to_str(ggml_type type) {257    return ggml_type_name(type);258}259 260static std::string var_to_str(ggml_op_pool pool) {261    switch (pool) {262        case GGML_OP_POOL_AVG:  return "avg";263        case GGML_OP_POOL_MAX:  return "max";264        default:                return std::to_string(pool);265    }266}267 268#define VAR_TO_STR(x) (#x "=" + var_to_str(x))269 270#define VARS_TO_STR1(a) VAR_TO_STR(a)271#define VARS_TO_STR2(a, b) VAR_TO_STR(a) + "," + VAR_TO_STR(b)272#define VARS_TO_STR3(a, b, c) VAR_TO_STR(a) + "," + VARS_TO_STR2(b, c)273#define VARS_TO_STR4(a, b, c, d) VAR_TO_STR(a) + "," + VARS_TO_STR3(b, c, d)274#define VARS_TO_STR5(a, b, c, d, e) VAR_TO_STR(a) + "," + VARS_TO_STR4(b, c, d, e)275#define VARS_TO_STR6(a, b, c, d, e, f) VAR_TO_STR(a) + "," + VARS_TO_STR5(b, c, d, e, f)276#define VARS_TO_STR7(a, b, c, d, e, f, g) VAR_TO_STR(a) + "," + VARS_TO_STR6(b, c, d, e, f, g)277#define VARS_TO_STR8(a, b, c, d, e, f, g, h) VAR_TO_STR(a) + "," + VARS_TO_STR7(b, c, d, e, f, g, h)278#define VARS_TO_STR9(a, b, c, d, e, f, g, h, i) VAR_TO_STR(a) + "," + VARS_TO_STR8(b, c, d, e, f, g, h, i)279#define VARS_TO_STR10(a, b, c, d, e, f, g, h, i, j) VAR_TO_STR(a) + "," + VARS_TO_STR9(b, c, d, e, f, g, h, i, j)280#define VARS_TO_STR11(a, b, c, d, e, f, g, h, i, j, k) VAR_TO_STR(a) + "," + VARS_TO_STR10(b, c, d, e, f, g, h, i, j, k)281#define VARS_TO_STR12(a, b, c, d, e, f, g, h, i, j, k, l) VAR_TO_STR(a) + "," + VARS_TO_STR11(b, c, d, e, f, g, h, i, j, k, l)282 283#ifdef GGML_USE_SYCL284static bool inline _isinf(float f) {285    return (*(uint32_t *)&f & 0x7fffffff) == 0x7f800000;286}287#else288static bool inline _isinf(float f) { return std::isinf(f); }289#endif290 291// accept FLT_MAX as infinity292static bool isinf_or_max(float f) {293    return _isinf(f) || f == FLT_MAX || f == -FLT_MAX;294}295 296static bool ggml_is_view_op(enum ggml_op op) {297    return op == GGML_OP_VIEW || op == GGML_OP_RESHAPE || op == GGML_OP_PERMUTE || op == GGML_OP_TRANSPOSE;298}299 300enum test_mode {301    MODE_TEST,302    MODE_PERF,303    MODE_GRAD,304};305 306struct test_case {307    virtual ~test_case() {}308 309    virtual std::string op_desc(ggml_tensor * t) {310        return ggml_op_desc(t);311    }312 313    virtual std::string vars() {314        return "";315    }316 317    virtual ggml_tensor * build_graph(ggml_context * ctx) = 0;318 319    virtual double max_nmse_err() {320        return 1e-7;321    }322 323    virtual double max_maa_err() {324        return 1e-4;325    }326 327    virtual float grad_eps() {328        return 1e-1f;329    }330 331    // If false, estimate gradient with 2 points, neglects 3rd order derivative and higher.332    // If true,  estimate gradient with 4 points, neglects 5th order derivative and higher.333    virtual bool grad_precise() {334        return false;335    }336 337    // Skip gradient checks if total number of gradients to be checked is larger than this (to speed up the tests).338    virtual int64_t grad_nmax() {339        return 10000;340    }341 342    // No effect if empty.343    // If not empty, skip all gradient checks where the numerical result does not match any of the values.344    // Needed for dealing with noncontinuous gradients (e.g. ReLU) where estimation using finite differences is unreliable.345    virtual std::vector<float> grad_expect() {346        return {};347    }348 349    virtual void initialize_tensors(ggml_context * ctx) {350        for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {351            init_tensor_uniform(t);352        }353    }354 355    virtual size_t op_size(ggml_tensor * t) {356        size_t size = ggml_nbytes(t);357        // add source tensors358        for (int i = 0; i < GGML_MAX_SRC; i++) {359            if (t->src[i] != NULL) {360                size += ggml_nbytes(t->src[i]);361            }362        }363        return size;364    }365 366    virtual uint64_t op_flops(ggml_tensor * t) {367        GGML_UNUSED(t);368        return 0;369    }370 371    ggml_cgraph * gf = nullptr;372    ggml_cgraph * gb = nullptr;373 374    static const int sentinel_size = 1024;375 376    test_mode mode;377 378    std::vector<ggml_tensor *> sentinels;379 380    void add_sentinel(ggml_context * ctx) {381        if (mode == MODE_PERF || mode == MODE_GRAD) {382            return;383        }384        ggml_tensor * sentinel = ::ggml_new_tensor_1d(ctx, GGML_TYPE_F32, sentinel_size);385        ggml_format_name(sentinel, "sent_%zu", sentinels.size());386        sentinels.push_back(sentinel);387    }388 389    // hijack ggml_new_tensor to add sentinels after each tensor to check for overflows in the backend390 391    ggml_tensor * ggml_new_tensor(ggml_context * ctx, ggml_type type, int n_dims, const int64_t * ne) {392        ggml_tensor * t = ::ggml_new_tensor(ctx, type, n_dims, ne);393        add_sentinel(ctx);394        return t;395    }396 397    ggml_tensor * ggml_new_tensor_1d(ggml_context * ctx, ggml_type type, int64_t ne0) {398        ggml_tensor * t = ::ggml_new_tensor_1d(ctx, type, ne0);399        add_sentinel(ctx);400        return t;401    }402 403    ggml_tensor * ggml_new_tensor_2d(ggml_context * ctx, ggml_type type, int64_t ne0, int64_t ne1) {404        ggml_tensor * t = ::ggml_new_tensor_2d(ctx, type, ne0, ne1);405        add_sentinel(ctx);406        return t;407    }408 409    ggml_tensor * ggml_new_tensor_3d(ggml_context * ctx, ggml_type type, int64_t ne0, int64_t ne1, int64_t ne2) {410        ggml_tensor * t = ::ggml_new_tensor_3d(ctx, type, ne0, ne1, ne2);411        add_sentinel(ctx);412        return t;413    }414 415    ggml_tensor * ggml_new_tensor_4d(ggml_context * ctx, ggml_type type, int64_t ne0, int64_t ne1, int64_t ne2, int64_t ne3) {416        ggml_tensor * t = ::ggml_new_tensor_4d(ctx, type, ne0, ne1, ne2, ne3);417        add_sentinel(ctx);418        return t;419    }420 421    bool eval(ggml_backend_t backend1, ggml_backend_t backend2, const char * op_name) {422        mode = MODE_TEST;423 424        ggml_init_params params = {425            /* .mem_size = */ ggml_tensor_overhead()*128 + ggml_graph_overhead(),426            /* .mem_base = */ NULL,427            /* .no_alloc = */ true,428        };429        ggml_context * ctx = ggml_init(params);430        GGML_ASSERT(ctx);431 432        gf = ggml_new_graph(ctx);433 434        // pre-graph sentinel435        add_sentinel(ctx);436 437        ggml_tensor * out = build_graph(ctx);438 439        if (op_name != nullptr && op_desc(out) != op_name) {440            //printf("  %s: skipping\n", op_desc(out).c_str());441            ggml_free(ctx);442            return true;443        }444 445        printf("  %s(%s): ", op_desc(out).c_str(), vars().c_str());446        fflush(stdout);447 448        // check if the backends support the ops449        bool supported = true;450        for (ggml_backend_t backend : {backend1, backend2}) {451            for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {452                if (!ggml_backend_supports_op(backend, t)) {453                    printf("not supported [%s] ", ggml_backend_name(backend));454                    supported = false;455                    break;456                }457            }458        }459        if (!supported) {460            printf("\n");461            ggml_free(ctx);462            return true;463        }464 465        // post-graph sentinel466        add_sentinel(ctx);467 468        // allocate469        ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors(ctx, backend1);470        if (buf == NULL) {471            printf("failed to allocate tensors [%s] ", ggml_backend_name(backend1));472            ggml_free(ctx);473            return false;474        }475 476        // build graph477        ggml_build_forward_expand(gf, out);478 479        // add sentinels as graph nodes so that they are checked in the callback480        for (ggml_tensor * sentinel : sentinels) {481            ggml_graph_add_node(gf, sentinel);482        }483 484        // randomize tensors485        initialize_tensors(ctx);486 487        // compare488        struct callback_userdata {489            bool   ok;490            double max_err;491            ggml_backend_t backend1;492            ggml_backend_t backend2;493        };494 495        callback_userdata ud {496            true,497            max_nmse_err(),498            backend1,499            backend2500        };501 502        auto callback = [](int index, ggml_tensor * t1, ggml_tensor * t2, void * user_data) -> bool {503            callback_userdata * ud = (callback_userdata *) user_data;504            const char * bn1 = ggml_backend_name(ud->backend1);505            const char * bn2 = ggml_backend_name(ud->backend2);506 507            if (t1->op == GGML_OP_NONE) {508                // sentinels must be unchanged509                std::vector<uint8_t> t1_data(ggml_nbytes(t1));510                std::vector<uint8_t> t2_data(ggml_nbytes(t2));511                ggml_backend_tensor_get(t1, t1_data.data(), 0, ggml_nbytes(t1));512                ggml_backend_tensor_get(t2, t2_data.data(), 0, ggml_nbytes(t2));513 514                if (memcmp(t1_data.data(), t2_data.data(), ggml_nbytes(t1)) != 0) {515                    printf("sentinel mismatch: %s ", t1->name);516                    ud->ok = false;517                    return true;518                }519            }520 521            std::vector<float> f1 = tensor_to_float(t1);522            std::vector<float> f2 = tensor_to_float(t2);523 524            for (size_t i = 0; i < f1.size(); i++) {525                // check for nans526                if (std::isnan(f1[i]) || std::isnan(f2[i])) {527                    printf("[%s] NaN at index %zu (%s=%f %s=%f) ", ggml_op_desc(t1), i, bn1, f1[i], bn2, f2[i]);528                    ud->ok = false;529                    return true;530                }531                // check for infs: both must be inf of the same sign, or both must be finite532                if (isinf_or_max(f1[i]) || isinf_or_max(f2[i])) {533                    if (isinf_or_max(f1[i]) && isinf_or_max(f2[i])) {534                        if (std::signbit(f1[i]) != std::signbit(f2[i])) {535                            printf("[%s] inf sign mismatch: %s=%f %s=%f ", ggml_op_desc(t1), bn1, f1[i], bn2, f2[i]);536                            ud->ok = false;537                            return true;538                        }539                    } else {540                        printf("[%s] inf mismatch: %s=%f %s=%f ", ggml_op_desc(t1), bn1, f1[i], bn2, f2[i]);541                        ud->ok = false;542                        return true;543                    }544                }545            }546 547            double err = nmse(f1.data(), f2.data(), f1.size());548            if (err > ud->max_err) {549                printf("[%s] NMSE = %.9f > %.9f ", ggml_op_desc(t1), err, ud->max_err);550                //for (int i = 0; i < (int) f1.size(); i++) {551                //    printf("%5d %9.6f %9.6f, diff = %9.6f\n", i, f1[i], f2[i], f1[i] - f2[i]);552                //}553                //printf("\n");554                //exit(1);555                ud->ok = false;556            }557            return true;558 559            GGML_UNUSED(index);560        };561 562        const bool cmp_ok = ggml_backend_compare_graph_backend(backend1, backend2, gf, callback, &ud);563 564        if (!cmp_ok) {565            printf("compare failed ");566        }567 568        ggml_backend_buffer_free(buf);569 570        ggml_free(ctx);571 572        if (ud.ok && cmp_ok) {573            printf("\033[1;32mOK\033[0m\n");574            return true;575        }576 577        printf("\033[1;31mFAIL\033[0m\n");578        return false;579    }580 581    bool eval_perf(ggml_backend_t backend, const char * op_name) {582        mode = MODE_PERF;583 584        static const size_t graph_nodes = 8192;585 586        ggml_init_params params = {587            /* .mem_size = */ ggml_tensor_overhead()*128 + ggml_graph_overhead_custom(graph_nodes, false),588            /* .mem_base = */ NULL,589            /* .no_alloc = */ true,590        };591        ggml_context * ctx = ggml_init(params);592        GGML_ASSERT(ctx);593 594        ggml_tensor * out = build_graph(ctx);595 596        if (op_name != nullptr && op_desc(out) != op_name) {597            //printf("  %s: skipping\n", op_desc(out).c_str());598            ggml_free(ctx);599            return true;600        }601 602        int len = printf("  %s(%s): ", op_desc(out).c_str(), vars().c_str());603        fflush(stdout);604 605        // check if backends support op606        if (!ggml_backend_supports_op(backend, out)) {607            printf("not supported\n");608            ggml_free(ctx);609            return true;610        }611 612        // align while also leaving some margin for variations in parameters613        int align = 8;614        int last = (len + align - 1) / align * align;615        if (last - len < 5) {616            last += align;617        }618        printf("%*s", last - len, "");619 620        // allocate621        ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors(ctx, backend);622        if (buf == NULL) {623            printf("failed to allocate tensors\n");624            ggml_free(ctx);625            return false;626        }627 628        // randomize tensors629        initialize_tensors(ctx);630 631        // build graph632        ggml_cgraph * gf = ggml_new_graph_custom(ctx, graph_nodes, false);633        ggml_build_forward_expand(gf, out);634 635        // warmup run636        ggml_backend_graph_compute(backend, gf);637 638        // determine number of runs639        int n_runs;640        bool is_cpu = ggml_backend_dev_type(ggml_backend_get_device(backend)) == GGML_BACKEND_DEVICE_TYPE_CPU;641        if (op_flops(out) > 0) {642            // based on flops643            const uint64_t GFLOP = 1000 * 1000 * 1000;644            const uint64_t target_flops_cpu =   8ULL * GFLOP;645            const uint64_t target_flops_gpu = 100ULL * GFLOP;646            uint64_t target_flops = is_cpu ? target_flops_cpu : target_flops_gpu;647            n_runs = std::min<int>(ggml_graph_size(gf) - ggml_graph_n_nodes(gf), target_flops / op_flops(out)) + 1;648        } else {649            // based on memory size650            const size_t GB = 1ULL << 30;651            const size_t target_size_cpu =  8 * GB;652            const size_t target_size_gpu = 32 * GB;653            size_t target_size = is_cpu ? target_size_cpu : target_size_gpu;654            n_runs = std::min<int>(ggml_graph_size(gf) - ggml_graph_n_nodes(gf), target_size / op_size(out)) + 1;655        }656 657        // duplicate the op658        for (int i = 1; i < n_runs; i++) {659            ggml_graph_add_node(gf, out);660        }661 662        // calculate memory663        size_t mem = n_runs * op_size(out);664        auto tensor_op_size = [](ggml_tensor * t) {665            size_t size = ggml_nbytes(t);666            // add source tensors667            for (int i = 0; i < GGML_MAX_SRC; i++) {668                if (t->src[i] != NULL) {669                    size += ggml_nbytes(t->src[i]);670                }671            }672            return size;673        };674        for (int i = 0; i < ggml_graph_n_nodes(gf); ++i) {675            if (ggml_is_view_op(ggml_graph_node(gf, i)->op) || ggml_graph_node(gf, i) == out) {676                continue;677            }678            mem += tensor_op_size(ggml_graph_node(gf, i));679        }680 681        // run682        int64_t total_time_us = 0;683        int64_t total_mem = 0;684        int total_runs = 0;685        do {686            int64_t start_time = ggml_time_us();687            ggml_backend_graph_compute(backend, gf);688            int64_t end_time = ggml_time_us();689 690            total_time_us += end_time - start_time;691            total_mem += mem;692            total_runs += n_runs;693        } while (total_time_us < 1000*1000); // run for at least 1 second694 695        printf("    %8d runs - %8.2f us/run - ",696            total_runs,697            (double)total_time_us / total_runs);698 699        if (op_flops(out) > 0) {700            double flops_per_sec = (op_flops(out) * total_runs) / (total_time_us / 1e6);701            auto format_flops = [](double flops) -> std::string {702                char buf[256];703                if (flops >= 1e12) {704                    snprintf(buf, sizeof(buf), "%6.2f TFLOP", flops / 1e12);705                } else if (flops >= 1e9) {706                    snprintf(buf, sizeof(buf), "%6.2f GFLOP", flops / 1e9);707                } else if (flops >= 1e6) {708                    snprintf(buf, sizeof(buf), "%6.2f MFLOP", flops / 1e6);709                } else {710                    snprintf(buf, sizeof(buf), "%6.2f KFLOP", flops / 1e3);711                }712                return buf;713            };714            printf("%s/run - \033[1;34m%sS\033[0m",715                format_flops(op_flops(out)).c_str(),716                format_flops(flops_per_sec).c_str());717 718        } else {719            printf("%8zu kB/run - \033[1;34m%7.2f GB/s\033[0m",720                op_size(out) / 1024,721                total_mem / (total_time_us / 1e6) / 1024.0 / 1024.0 / 1024.0);722        }723        printf("\n");724 725        ggml_backend_buffer_free(buf);726 727        ggml_free(ctx);728 729        return true;730    }731 732    bool eval_grad(ggml_backend_t backend, const char * op_name) {733        mode = MODE_GRAD;734        const std::vector<float> expect = grad_expect();735 736        ggml_init_params params = {737            /* .mem_size = */ ggml_tensor_overhead()*128 + 2*ggml_graph_overhead_custom(GGML_DEFAULT_GRAPH_SIZE, true),738            /* .mem_base = */ NULL,739            /* .no_alloc = */ true,740        };741        ggml_context * ctx = ggml_init(params);742        GGML_ASSERT(ctx);743 744        gf = ggml_new_graph_custom(ctx, GGML_DEFAULT_GRAPH_SIZE, true);745        gb = ggml_new_graph_custom(ctx, GGML_DEFAULT_GRAPH_SIZE, true);746 747        ggml_tensor * out = build_graph(ctx);748 749        if ((op_name != nullptr && op_desc(out) != op_name) || out->op == GGML_OP_OPT_STEP_ADAMW) {750            //printf("  %s: skipping\n", op_desc(out).c_str());751            ggml_free(ctx);752            return true;753        }754 755        printf("  %s(%s): ", op_desc(out).c_str(), vars().c_str());756        fflush(stdout);757 758        if (out->type != GGML_TYPE_F32) {759            ggml_free(ctx);760            printf("not supported [%s->type != FP32]\n", out->name);761            return true;762        }763 764        // check if the backend supports the ops765        bool supported = true;766        bool any_params = false;767        for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {768            if (!ggml_backend_supports_op(backend, t)) {769                printf("not supported [%s] ", ggml_backend_name(backend));770                supported = false;771                break;772            }773            if ((t->flags & GGML_TENSOR_FLAG_PARAM)) {774                any_params = true;775                if (t->type != GGML_TYPE_F32) {776                    printf("not supported [%s->type != FP32] ", t->name);777                    supported = false;778                    break;779                }780            }781        }782        if (!any_params) {783            printf("not supported [%s] \n", op_desc(out).c_str());784            supported = false;785        }786        if (!supported) {787            printf("\n");788            ggml_free(ctx);789            return true;790        }791 792        int64_t ngrads = 0;793        for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {794            if (t->flags & GGML_TENSOR_FLAG_PARAM) {795                ngrads += ggml_nelements(t);796            }797        }798        if (ngrads > grad_nmax()) {799            printf("skipping large tensors for speed \n");800            ggml_free(ctx);801            return true;802        }803 804 805        if (!ggml_is_scalar(out)) {806            out = ggml_sum(ctx, out);807            ggml_set_name(out, "sum_of_out");808        }809        ggml_set_loss(out);810 811        ggml_build_forward_expand(gf, out);812        ggml_graph_cpy(gf, gb);813        ggml_build_backward_expand(ctx, ctx, gb, false);814        if (expect.size() != 1 || expect[0] != 0.0f) {815            GGML_ASSERT(ggml_graph_n_nodes(gb) > ggml_graph_n_nodes(gf));816            for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {817                GGML_ASSERT(!(t->flags & GGML_TENSOR_FLAG_PARAM) || ggml_graph_get_grad(gb, t)->op != GGML_OP_NONE);818            }819        }820 821        for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {822            if (!ggml_backend_supports_op(backend, t)) {823                printf("not supported [%s] ", ggml_backend_name(backend));824                supported = false;825                break;826            }827            if ((t->flags & GGML_TENSOR_FLAG_PARAM) && t->type != GGML_TYPE_F32) {828                printf("not supported [%s->type != FP32] ", t->name);829                supported = false;830                break;831            }832        }833        if (!supported) {834            printf("\n");835            ggml_free(ctx);836            return true;837        }838 839        // allocate840        ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors(ctx, backend);841        if (buf == NULL) {842            printf("failed to allocate tensors [%s] ", ggml_backend_name(backend));843            ggml_free(ctx);844            return false;845        }846 847 848        initialize_tensors(ctx); // Randomizes all tensors (including gradients).849        ggml_graph_reset(gb);    // Sets gradients to 1 if loss, 0 otherwise.850 851        ggml_backend_graph_compute(backend, gf);852        ggml_backend_graph_compute(backend, gb);853 854        bool ok = true;855        for (struct ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {856            if (!(t->flags & GGML_TENSOR_FLAG_PARAM)) {857                continue;858            }859 860            const char * bn = ggml_backend_name(backend);861            const int64_t ne = ggml_nelements(t);862 863            std::vector<float> ga;864            struct ggml_tensor * grad = ggml_graph_get_grad(gb, t);865            if (grad) {866                ga = tensor_to_float(grad);867            } else {868                ga.resize(ne); // default value is 0.0f869            }870 871            for (int64_t i = 0; i < ne; ++i) { // gradient algebraic872                // check for nans873                if (!std::isfinite(ga[i])) {874                    printf("[%s] nonfinite gradient at index %" PRId64 " (%s=%f) ", ggml_op_desc(t), i, bn, ga[i]);875                    ok = false;876                    break;877                }878            }879            if (!ok) {880                break;881            }882 883            std::vector<float> gn(ne); // gradient numeric884            GGML_ASSERT(ga.size() == gn.size());885 886            std::vector<float> x0 = tensor_to_float(t); // original t data887            GGML_ASSERT(ggml_is_scalar(out));888            GGML_ASSERT(out->type == GGML_TYPE_F32);889 890            const float eps = grad_eps();891            for (int64_t i = 0; i < ne; ++i) {892                const float xiu  = x0[i] + 1.0f*eps; // x, index i, up893                const float xiuh = x0[i] + 0.5f*eps; // x, index i, up half894                const float xidh = x0[i] - 0.5f*eps; // x, index i, down half895                const float xid  = x0[i] - 1.0f*eps; // x, index i, down896 897                float fu, fuh, fdh, fd; // output values for xiu, xiuh, xid, xidh898 899                ggml_backend_tensor_set(t, &xiu, i*sizeof(float), sizeof(float));900                ggml_backend_graph_compute(backend, gf);901                ggml_backend_tensor_get(out, &fu, 0, ggml_nbytes(out));902 903                ggml_backend_tensor_set(t, &xid, i*sizeof(float), sizeof(float));904                ggml_backend_graph_compute(backend, gf);905                ggml_backend_tensor_get(out, &fd, 0, ggml_nbytes(out));906 907                if (grad_precise()) {908                    ggml_backend_tensor_set(t, &xiuh, i*sizeof(float), sizeof(float));909                    ggml_backend_graph_compute(backend, gf);910                    ggml_backend_tensor_get(out, &fuh, 0, ggml_nbytes(out));911 912                    ggml_backend_tensor_set(t, &xidh, i*sizeof(float), sizeof(float));913                    ggml_backend_graph_compute(backend, gf);914                    ggml_backend_tensor_get(out, &fdh, 0, ggml_nbytes(out));915 916                    gn[i] = (8.0*(double)fuh + (double)fd - (8.0*(double)fdh + (double)fu)) / (6.0*(double)eps);917                } else {918                    gn[i] = (fu - fd) / (2.0f*eps);919                }920 921                ggml_backend_tensor_set(t, x0.data(), 0, ggml_nbytes(t));922            }923 924            const double err = mean_abs_asymm(gn.data(), ga.data(), gn.size(), expect);925            if (err > max_maa_err()) {926                printf("[%s] MAA = %.9f > %.9f ", ggml_op_desc(t), err, max_maa_err());927                ok = false;928                break;929            }930            if (!ok) {931                break;932            }933        }934 935        if (!ok) {936            printf("compare failed ");937        }938 939        ggml_backend_buffer_free(buf);940 941        ggml_free(ctx);942 943        if (ok) {944            printf("\033[1;32mOK\033[0m\n");945            return true;946        }947 948        printf("\033[1;31mFAIL\033[0m\n");949        return false;950    }951};952 953 954// ###################################955// ## Section 2: GGML Op Defintions ##956// ###################################957 958 959// The following is an example showing the bare minimum for creating a test for a GGML op.960 961// GGML_OP_EXAMPLE962struct test_example : public test_case {963    // Always define these 2 or variants thereof:964    const ggml_type type; // The type of the input tensors.965    const std::array<int64_t, 4> ne; // The shape of the input tensors.966    // For some ops it's necessary to define multiple types or shapes for the inputs.967    // Or they may need additional parameters.968 969    // Put all parameters needed to fully define the test into one of the VARS_TO_STR macros.970    // In most cases these are just the properties of the struct that you defined above.971    // This is needed for info prints.972    std::string vars() override {973        return VARS_TO_STR2(type, ne);974    }975 976    // Define a constructor for the struct.977    // In most cases it will be sufficient to have the same arguments as the struct has properties978    // and just use initializer lists.979    test_example(ggml_type type = GGML_TYPE_F32,980            std::array<int64_t, 4> ne = {10, 5, 4, 3})981        : type(type), ne(ne) {}982 983    // Define how a simple GGML compute graph can be constructed for the new GGML op.984    ggml_tensor * build_graph(ggml_context * ctx) override {985        // Step 1: create input tensors that don't depend on any other tensors:986        ggml_tensor * a = ggml_new_tensor(ctx, type, 4, ne.data());987        ggml_set_name(a, "a"); // Setting names is optional but it's useful for debugging.988 989        ggml_tensor * b = ggml_new_tensor(ctx, type, 4, ne.data());990        ggml_set_name(b, "b");991 992        // Step 2: use the op that you want to test in the GGML compute graph.993        ggml_tensor * out = ggml_add(ctx, a, b); // For this example we're just doing a simple addition.994        ggml_set_name(out, "out");995 996        // Step 3: return the output tensor.997        return out;998    }999    // In order to also check the gradients for your op, add calls like ggml_set_param(ctx, a)1000    // immediately after you create the tensors.1001    // This is optional and only makes sense if a backward pass has actually been implemented for the new op.1002};1003 1004 1005// GGML_OP_UNARY1006struct test_unary : public test_case {1007    const ggml_unary_op op;1008    const ggml_type type;1009    const std::array<int64_t, 4> ne_a;1010    int v; // view (1 : non-contiguous a)1011 1012    std::string vars() override {1013        return VARS_TO_STR3(type, ne_a, v);1014    }1015 1016    test_unary(ggml_unary_op op,1017            ggml_type type = GGML_TYPE_F32,1018            std::array<int64_t, 4> ne_a = {128, 2, 2, 2},1019            int v = 0)1020        : op(op), type(type), ne_a(ne_a), v(v) {}1021 1022    ggml_tensor * build_graph(ggml_context * ctx) override {1023        const bool grad_supported = op == GGML_UNARY_OP_ABS || op == GGML_UNARY_OP_SGN || op == GGML_UNARY_OP_NEG ||1024            op == GGML_UNARY_OP_STEP || op == GGML_UNARY_OP_RELU || op == GGML_UNARY_OP_SILU;1025 1026        ggml_tensor * a;1027        if (v & 1) {1028            auto ne = ne_a; ne[0] *= 3;1029            a = ggml_new_tensor(ctx, type, 4, ne.data());1030            if (grad_supported) {1031                ggml_set_param(ctx, a);1032            }1033            ggml_set_name(a, "a");1034 1035            a = ggml_view_4d(ctx, a, ne_a[0], ne_a[1], ne_a[2], ne_a[3], a->nb[1], a->nb[2], a->nb[3], 0);1036            ggml_set_name(a, "view_of_a");1037        } else {1038            a = ggml_new_tensor(ctx, type, 4, ne_a.data());1039            if (grad_supported) {1040                ggml_set_param(ctx, a);1041            }1042            ggml_set_name(a, "a");1043        }1044 1045        ggml_tensor * out = ggml_unary(ctx, a, op);1046        ggml_set_name(out, "out");1047 1048        return out;1049    }1050 1051    void initialize_tensors(ggml_context * ctx) override {1052        for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {1053            // test extended range of values to check for NaNs in GELU1054            init_tensor_uniform(t, -150.f, 150.f);1055        }1056    }1057 1058    float grad_eps() override {1059        return 15.0f;1060    }1061 1062    std::vector<float> grad_expect() override {1063        if (op == GGML_UNARY_OP_ABS) {1064            return {-1.0f, 1.0f};1065        }1066        if (op == GGML_UNARY_OP_SGN || op == GGML_UNARY_OP_STEP) {1067            return {0.0f};1068        }1069        if (op == GGML_UNARY_OP_RELU) {1070            return {0.0f, 1.0f};1071        }1072        return {};1073    }1074 1075};1076 1077// GGML_OP_GET_ROWS1078struct test_get_rows : public test_case {1079    const ggml_type type;1080    const int n; // cols1081    const int m; // rows1082    const int r; // rows to get1083    const int b; // batch size1084    const bool v; // view (non-contiguous src1)1085 1086    std::string vars() override {1087        return VARS_TO_STR6(type, n, m, r, b, v);1088    }1089 1090    test_get_rows(ggml_type type = GGML_TYPE_F32, int n = 10, int m = 5, int r = 3, int b = 1, bool v = false)1091        : type(type), n(n), m(m), r(r), b(b), v(v) {}1092 1093    ggml_tensor * build_graph(ggml_context * ctx) override {1094        ggml_tensor * in = ggml_new_tensor_3d(ctx, type, n, m, b);1095        ggml_set_name(in, "in");1096 1097        ggml_tensor * rows = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, r, b);1098        ggml_set_name(rows, "rows");1099        if (v) {1100            rows = ggml_view_2d(ctx, rows, r/2, b, rows->nb[1], 0);1101            ggml_set_name(rows, "view_of_rows");1102        }1103 1104        const bool grad_supported = ggml_is_matrix(in) && ggml_is_vector(rows);1105        if (grad_supported) {1106            ggml_set_param(ctx, in);1107            // rows is a constant input -> no gradients1108        }1109 1110        ggml_tensor * out = ggml_get_rows(ctx, in, rows);1111        ggml_set_name(out, "out");1112 1113        return out;1114    }1115 1116    void initialize_tensors(ggml_context * ctx) override {1117        for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {1118            if (t->type == GGML_TYPE_I32) {1119                if (ggml_is_view_op(t->op)) { continue; }1120                // rows1121                std::vector<int> data(r*b);1122                for (int i = 0; i < r*b; i++) {1123                    data[i] = rand() % m;1124                }1125                ggml_backend_tensor_set(t, data.data(), 0, r * b * sizeof(int));1126            } else {1127                init_tensor_uniform(t);1128            }1129        }1130    }1131};1132 1133// GGML_OP_GET_ROWS_BACK1134struct test_get_rows_back : public test_case {1135    const ggml_type type;1136    const int n; // cols1137    const int m; // rows1138    const int r; // rows to get1139    const int b; // batch size1140    const bool v; // view (non-contiguous src1)1141 1142    std::string vars() override {1143        return VARS_TO_STR6(type, n, m, r, b, v);1144    }1145 1146    test_get_rows_back(ggml_type type = GGML_TYPE_F32, int n = 10, int m = 5, int r = 3, int b = 1, bool v = false)1147        : type(type), n(n), m(m), r(r), b(b), v(v) {}1148 1149    ggml_tensor * build_graph(ggml_context * ctx) override {1150        ggml_tensor * in_forward = ggml_new_tensor_3d(ctx, type, n, m, b);1151        ggml_set_name(in_forward, "in_forward");1152 1153        ggml_tensor * rows = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, r, b);1154        ggml_set_name(rows, "rows");1155        if (v) {1156            rows = ggml_view_2d(ctx, rows, r/2, b, rows->nb[1], 0);1157            ggml_set_name(rows, "view_of_rows");1158        }1159 1160        ggml_tensor * grad = ggml_new_tensor_3d(ctx, type, n, r, b);1161        ggml_set_name(grad, "grad");1162 1163        ggml_tensor * out = ggml_get_rows_back(ctx, grad, rows, in_forward);1164        ggml_set_name(out, "out");1165 1166        return out;1167    }1168 1169    void initialize_tensors(ggml_context * ctx) override {1170        for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {1171            if (t->type == GGML_TYPE_I32) {1172                if (ggml_is_view_op(t->op)) { continue; }1173                // rows1174                std::vector<int> data(r*b);1175                for (int i = 0; i < r*b; i++) {1176                    data[i] = rand() % m;1177                }1178                ggml_backend_tensor_set(t, data.data(), 0, r * b * sizeof(int));1179            } else {1180                init_tensor_uniform(t);1181            }1182        }1183    }1184};1185 1186// GGML_OP_ARGMAX1187struct test_argmax : public test_case {1188    const ggml_type type;1189    const std::array<int64_t, 4> ne;1190 1191    std::string vars() override {1192        return VARS_TO_STR2(type, ne);1193    }1194 1195    test_argmax(ggml_type type = GGML_TYPE_F32,1196            std::array<int64_t, 4> ne = {10, 100, 1, 1})1197        : type(type), ne(ne) {}1198 1199    ggml_tensor * build_graph(ggml_context * ctx) override {1200        ggml_tensor * a = ggml_new_tensor(ctx, type, 4, ne.data());

Showing the first 1,200 of 4517 lines. Download the file for the rest.