KBaba7/llama.cpp
0
1// This file defines tests for various GGML ops and backends.2// For the forward pass it asserts that the results of multiple backends computing the same GGML ops are consistent.3// For the backward pass it asserts that the gradients from backpropagation are consistent4// with the gradients obtained via the method of finite differences ("grad" mode, this is optional).5// It is also possible to check the performance ("perf" mode).6//7// this file has three sections: Section 1 does general setup, section 2 defines the GGML ops to be tested,8// and section 3 defines which tests to run.9// Quick start for adding a new GGML op: Go to section 2 and create a struct that inherits from test_case,10// then go to section 3 and add an instantiation of your struct.11 12 13// ##############################14// ## Section 1: General Setup ##15// ##############################16 17 18#include <ggml.h>19#include <ggml-alloc.h>20#include <ggml-backend.h>21 22#include <algorithm>23#include <array>24#include <cfloat>25#include <cstdint>26#include <cstring>27#include <cinttypes>28#include <memory>29#include <random>30#include <stdio.h>31#include <stdlib.h>32#include <string>33#include <thread>34#include <future>35#include <vector>36 37static void init_tensor_uniform(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) {38 size_t nels = ggml_nelements(tensor);39 std::vector<float> data(nels);40 {41 // parallel initialization42 static const size_t n_threads = std::thread::hardware_concurrency();43 // static RNG initialization (revisit if n_threads stops being constant)44 static std::vector<std::default_random_engine> generators = []() {45 std::random_device rd;46 std::vector<std::default_random_engine> vec;47 vec.reserve(n_threads);48 //for (size_t i = 0; i < n_threads; i++) { vec.emplace_back(1234 + i); } // fixed seed49 for (size_t i = 0; i < n_threads; i++) { vec.emplace_back(rd()); }50 return vec;51 }();52 53 auto init_thread = [&](size_t ith, size_t start, size_t end) {54 std::uniform_real_distribution<float> distribution(min, max);55 auto & gen = generators[ith];56 for (size_t i = start; i < end; i++) {57 data[i] = distribution(gen);58 }59 };60 61 std::vector<std::future<void>> tasks;62 tasks.reserve(n_threads);63 for (size_t i = 0; i < n_threads; i++) {64 size_t start = i*nels/n_threads;65 size_t end = (i+1)*nels/n_threads;66 tasks.push_back(std::async(std::launch::async, init_thread, i, start, end));67 }68 for (auto & t : tasks) {69 t.get();70 }71 }72 73 if (tensor->type == GGML_TYPE_F32 || tensor->type == GGML_TYPE_I32) {74 ggml_backend_tensor_set(tensor, data.data(), 0, nels * sizeof(float));75 } else if (ggml_is_quantized(tensor->type) || tensor->type == GGML_TYPE_F16 || tensor->type == GGML_TYPE_BF16) {76 GGML_ASSERT(nels % ggml_blck_size(tensor->type) == 0);77 78 // dummy importance matrix79 std::vector<float> imatrix(tensor->ne[0], 1.0f);80 const float * im = imatrix.data();81 if (!ggml_quantize_requires_imatrix(tensor->type)) {82 // when the imatrix is optional, we want to test both quantization with and without imatrix83 // use one of the random numbers to decide84 if (data[0] > 0.5f*(min + max)) {85 im = nullptr;86 }87 }88 89 std::vector<uint8_t> dataq(ggml_row_size(tensor->type, nels));90 {91 // parallel quantization by block92 size_t blck_size = ggml_blck_size(tensor->type);93 size_t n_blocks = nels / blck_size;94 95 auto quantize_thread = [&](size_t start, size_t end) {96 ggml_quantize_chunk(tensor->type, data.data(), dataq.data(),97 start * blck_size, end - start, blck_size, im);98 };99 100 const size_t min_blocks_per_thread = 1;101 const size_t n_threads = std::min<size_t>(std::thread::hardware_concurrency()/2,102 std::max<size_t>(1, n_blocks / min_blocks_per_thread));103 std::vector<std::future<void>> tasks;104 tasks.reserve(n_threads);105 for (size_t i = 0; i < n_threads; i++) {106 size_t start = i*n_blocks/n_threads;107 size_t end = (i+1)*n_blocks/n_threads;108 tasks.push_back(std::async(std::launch::async, quantize_thread, start, end));109 }110 for (auto & t : tasks) {111 t.get();112 }113 }114 ggml_backend_tensor_set(tensor, dataq.data(), 0, dataq.size());115 } else if (tensor->type == GGML_TYPE_I8 || tensor->type == GGML_TYPE_I16 || tensor->type == GGML_TYPE_I32) {116 // This is going to create some weird integers though.117 ggml_backend_tensor_set(tensor, data.data(), 0, ggml_nbytes(tensor));118 } else if (tensor->type == GGML_TYPE_I64) {119 // Integers with a size of 8 bytes can be set by mirroring the float data, the specific values are again not really meaningful.120 const size_t nbytes_half = ggml_nbytes(tensor)/2;121 ggml_backend_tensor_set(tensor, data.data(), 0*nbytes_half, nbytes_half);122 ggml_backend_tensor_set(tensor, data.data(), 1*nbytes_half, nbytes_half);123 } else {124 GGML_ABORT("fatal error");125 }126}127 128static std::vector<float> tensor_to_float(const ggml_tensor * t) {129 std::vector<float> tv;130 tv.reserve(ggml_nelements(t));131 132 std::vector<uint8_t> buf(ggml_nbytes(t));133 ggml_backend_tensor_get(t, buf.data(), 0, ggml_nbytes(t));134 135 const auto * tt = ggml_get_type_traits(t->type);136 size_t bs = ggml_blck_size(t->type);137 std::vector<float> vq(ggml_blck_size(t->type));138 bool quantized = ggml_is_quantized(t->type);139 140 // access elements by index to avoid gaps in views141 for (int64_t i3 = 0; i3 < t->ne[3]; i3++) {142 for (int64_t i2 = 0; i2 < t->ne[2]; i2++) {143 for (int64_t i1 = 0; i1 < t->ne[1]; i1++) {144 for (int64_t i0 = 0; i0 < t->ne[0]; i0 += bs) {145 size_t i = i3*t->nb[3] + i2*t->nb[2] + i1*t->nb[1] + i0/bs*t->nb[0];146 if (t->type == GGML_TYPE_F16) {147 tv.push_back(ggml_fp16_to_fp32(*(ggml_fp16_t*)&buf[i]));148 } else if (t->type == GGML_TYPE_BF16) {149 tv.push_back(ggml_bf16_to_fp32(*(ggml_bf16_t*)&buf[i]));150 } else if (t->type == GGML_TYPE_F32) {151 tv.push_back(*(float *) &buf[i]);152 } else if (t->type == GGML_TYPE_I64) {153 tv.push_back((float)*(int64_t *) &buf[i]);154 } else if (t->type == GGML_TYPE_I32) {155 tv.push_back((float)*(int32_t *) &buf[i]);156 } else if (t->type == GGML_TYPE_I16) {157 tv.push_back((float)*(int16_t *) &buf[i]);158 } else if (t->type == GGML_TYPE_I8) {159 tv.push_back((float)*(int8_t *) &buf[i]);160 } else if (quantized) {161 tt->to_float(&buf[i], vq.data(), bs);162 tv.insert(tv.end(), vq.begin(), vq.end());163 } else {164 GGML_ABORT("fatal error");165 }166 }167 }168 }169 }170 171 return tv;172}173 174// normalized mean squared error = mse(a, b) / mse(a, 0)175static double nmse(const float * a, const float * b, size_t n) {176 double mse_a_b = 0.0;177 double mse_a_0 = 0.0;178 179 for (size_t i = 0; i < n; i++) {180 float a_i = a[i];181 float b_i = b[i];182 183 mse_a_b += (a_i - b_i) * (a_i - b_i);184 mse_a_0 += a_i * a_i;185 }186 187 return mse_a_b / mse_a_0;188}189 190// maximum absolute asymmetry between a and b191// asymmetry: (a - b) / (a + b)192// This is more stable than relative error if one of the values fluctuates towards zero.193// n: number of values to compare.194// expected_vals: optional vector of expected values for a. If expected_vals is not empty, filter out all comparisons where195// a does not match any of the expected values. Needed for noncontinuous gradients where the numerical calculation can fail.196static double mean_abs_asymm(const float * a, const float * b, const size_t n, const std::vector<float> & expected_vals) {197 double sum = 0.0f;198 199 size_t nvalid = 0;200 for (size_t i = 0; i < n; i++) {201 if (!expected_vals.empty()) {202 bool matches_any = false;203 for (const float & ev : expected_vals) {204 if (fabsf(a[i] - ev) < 1e-3f) {205 matches_any = true;206 break;207 }208 }209 if (!matches_any) {210 continue;211 }212 }213 214 const float asymm = (a[i] - b[i]) / (a[i] + b[i]);215 216 sum += fabsf(asymm);217 nvalid++;218 }219 220 return sum/nvalid;221}222 223// utils for printing the variables of the test cases224 225template<typename T>226static std::string var_to_str(const T & x) {227 return std::to_string(x);228}229 230template<typename T, size_t N>231static std::string var_to_str(const T (&x)[N]) {232 std::string s = "[";233 for (size_t i = 0; i < N; i++) {234 if (i > 0) {235 s += ",";236 }237 s += var_to_str(x[i]);238 }239 s += "]";240 return s;241}242 243template<typename T, size_t N>244static std::string var_to_str(const std::array<T, N> & x) {245 std::string s = "[";246 for (size_t i = 0; i < N; i++) {247 if (i > 0) {248 s += ",";249 }250 s += var_to_str(x[i]);251 }252 s += "]";253 return s;254}255 256static std::string var_to_str(ggml_type type) {257 return ggml_type_name(type);258}259 260static std::string var_to_str(ggml_op_pool pool) {261 switch (pool) {262 case GGML_OP_POOL_AVG: return "avg";263 case GGML_OP_POOL_MAX: return "max";264 default: return std::to_string(pool);265 }266}267 268#define VAR_TO_STR(x) (#x "=" + var_to_str(x))269 270#define VARS_TO_STR1(a) VAR_TO_STR(a)271#define VARS_TO_STR2(a, b) VAR_TO_STR(a) + "," + VAR_TO_STR(b)272#define VARS_TO_STR3(a, b, c) VAR_TO_STR(a) + "," + VARS_TO_STR2(b, c)273#define VARS_TO_STR4(a, b, c, d) VAR_TO_STR(a) + "," + VARS_TO_STR3(b, c, d)274#define VARS_TO_STR5(a, b, c, d, e) VAR_TO_STR(a) + "," + VARS_TO_STR4(b, c, d, e)275#define VARS_TO_STR6(a, b, c, d, e, f) VAR_TO_STR(a) + "," + VARS_TO_STR5(b, c, d, e, f)276#define VARS_TO_STR7(a, b, c, d, e, f, g) VAR_TO_STR(a) + "," + VARS_TO_STR6(b, c, d, e, f, g)277#define VARS_TO_STR8(a, b, c, d, e, f, g, h) VAR_TO_STR(a) + "," + VARS_TO_STR7(b, c, d, e, f, g, h)278#define VARS_TO_STR9(a, b, c, d, e, f, g, h, i) VAR_TO_STR(a) + "," + VARS_TO_STR8(b, c, d, e, f, g, h, i)279#define VARS_TO_STR10(a, b, c, d, e, f, g, h, i, j) VAR_TO_STR(a) + "," + VARS_TO_STR9(b, c, d, e, f, g, h, i, j)280#define VARS_TO_STR11(a, b, c, d, e, f, g, h, i, j, k) VAR_TO_STR(a) + "," + VARS_TO_STR10(b, c, d, e, f, g, h, i, j, k)281#define VARS_TO_STR12(a, b, c, d, e, f, g, h, i, j, k, l) VAR_TO_STR(a) + "," + VARS_TO_STR11(b, c, d, e, f, g, h, i, j, k, l)282 283#ifdef GGML_USE_SYCL284static bool inline _isinf(float f) {285 return (*(uint32_t *)&f & 0x7fffffff) == 0x7f800000;286}287#else288static bool inline _isinf(float f) { return std::isinf(f); }289#endif290 291// accept FLT_MAX as infinity292static bool isinf_or_max(float f) {293 return _isinf(f) || f == FLT_MAX || f == -FLT_MAX;294}295 296static bool ggml_is_view_op(enum ggml_op op) {297 return op == GGML_OP_VIEW || op == GGML_OP_RESHAPE || op == GGML_OP_PERMUTE || op == GGML_OP_TRANSPOSE;298}299 300enum test_mode {301 MODE_TEST,302 MODE_PERF,303 MODE_GRAD,304};305 306struct test_case {307 virtual ~test_case() {}308 309 virtual std::string op_desc(ggml_tensor * t) {310 return ggml_op_desc(t);311 }312 313 virtual std::string vars() {314 return "";315 }316 317 virtual ggml_tensor * build_graph(ggml_context * ctx) = 0;318 319 virtual double max_nmse_err() {320 return 1e-7;321 }322 323 virtual double max_maa_err() {324 return 1e-4;325 }326 327 virtual float grad_eps() {328 return 1e-1f;329 }330 331 // If false, estimate gradient with 2 points, neglects 3rd order derivative and higher.332 // If true, estimate gradient with 4 points, neglects 5th order derivative and higher.333 virtual bool grad_precise() {334 return false;335 }336 337 // Skip gradient checks if total number of gradients to be checked is larger than this (to speed up the tests).338 virtual int64_t grad_nmax() {339 return 10000;340 }341 342 // No effect if empty.343 // If not empty, skip all gradient checks where the numerical result does not match any of the values.344 // Needed for dealing with noncontinuous gradients (e.g. ReLU) where estimation using finite differences is unreliable.345 virtual std::vector<float> grad_expect() {346 return {};347 }348 349 virtual void initialize_tensors(ggml_context * ctx) {350 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {351 init_tensor_uniform(t);352 }353 }354 355 virtual size_t op_size(ggml_tensor * t) {356 size_t size = ggml_nbytes(t);357 // add source tensors358 for (int i = 0; i < GGML_MAX_SRC; i++) {359 if (t->src[i] != NULL) {360 size += ggml_nbytes(t->src[i]);361 }362 }363 return size;364 }365 366 virtual uint64_t op_flops(ggml_tensor * t) {367 GGML_UNUSED(t);368 return 0;369 }370 371 ggml_cgraph * gf = nullptr;372 ggml_cgraph * gb = nullptr;373 374 static const int sentinel_size = 1024;375 376 test_mode mode;377 378 std::vector<ggml_tensor *> sentinels;379 380 void add_sentinel(ggml_context * ctx) {381 if (mode == MODE_PERF || mode == MODE_GRAD) {382 return;383 }384 ggml_tensor * sentinel = ::ggml_new_tensor_1d(ctx, GGML_TYPE_F32, sentinel_size);385 ggml_format_name(sentinel, "sent_%zu", sentinels.size());386 sentinels.push_back(sentinel);387 }388 389 // hijack ggml_new_tensor to add sentinels after each tensor to check for overflows in the backend390 391 ggml_tensor * ggml_new_tensor(ggml_context * ctx, ggml_type type, int n_dims, const int64_t * ne) {392 ggml_tensor * t = ::ggml_new_tensor(ctx, type, n_dims, ne);393 add_sentinel(ctx);394 return t;395 }396 397 ggml_tensor * ggml_new_tensor_1d(ggml_context * ctx, ggml_type type, int64_t ne0) {398 ggml_tensor * t = ::ggml_new_tensor_1d(ctx, type, ne0);399 add_sentinel(ctx);400 return t;401 }402 403 ggml_tensor * ggml_new_tensor_2d(ggml_context * ctx, ggml_type type, int64_t ne0, int64_t ne1) {404 ggml_tensor * t = ::ggml_new_tensor_2d(ctx, type, ne0, ne1);405 add_sentinel(ctx);406 return t;407 }408 409 ggml_tensor * ggml_new_tensor_3d(ggml_context * ctx, ggml_type type, int64_t ne0, int64_t ne1, int64_t ne2) {410 ggml_tensor * t = ::ggml_new_tensor_3d(ctx, type, ne0, ne1, ne2);411 add_sentinel(ctx);412 return t;413 }414 415 ggml_tensor * ggml_new_tensor_4d(ggml_context * ctx, ggml_type type, int64_t ne0, int64_t ne1, int64_t ne2, int64_t ne3) {416 ggml_tensor * t = ::ggml_new_tensor_4d(ctx, type, ne0, ne1, ne2, ne3);417 add_sentinel(ctx);418 return t;419 }420 421 bool eval(ggml_backend_t backend1, ggml_backend_t backend2, const char * op_name) {422 mode = MODE_TEST;423 424 ggml_init_params params = {425 /* .mem_size = */ ggml_tensor_overhead()*128 + ggml_graph_overhead(),426 /* .mem_base = */ NULL,427 /* .no_alloc = */ true,428 };429 ggml_context * ctx = ggml_init(params);430 GGML_ASSERT(ctx);431 432 gf = ggml_new_graph(ctx);433 434 // pre-graph sentinel435 add_sentinel(ctx);436 437 ggml_tensor * out = build_graph(ctx);438 439 if (op_name != nullptr && op_desc(out) != op_name) {440 //printf(" %s: skipping\n", op_desc(out).c_str());441 ggml_free(ctx);442 return true;443 }444 445 printf(" %s(%s): ", op_desc(out).c_str(), vars().c_str());446 fflush(stdout);447 448 // check if the backends support the ops449 bool supported = true;450 for (ggml_backend_t backend : {backend1, backend2}) {451 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {452 if (!ggml_backend_supports_op(backend, t)) {453 printf("not supported [%s] ", ggml_backend_name(backend));454 supported = false;455 break;456 }457 }458 }459 if (!supported) {460 printf("\n");461 ggml_free(ctx);462 return true;463 }464 465 // post-graph sentinel466 add_sentinel(ctx);467 468 // allocate469 ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors(ctx, backend1);470 if (buf == NULL) {471 printf("failed to allocate tensors [%s] ", ggml_backend_name(backend1));472 ggml_free(ctx);473 return false;474 }475 476 // build graph477 ggml_build_forward_expand(gf, out);478 479 // add sentinels as graph nodes so that they are checked in the callback480 for (ggml_tensor * sentinel : sentinels) {481 ggml_graph_add_node(gf, sentinel);482 }483 484 // randomize tensors485 initialize_tensors(ctx);486 487 // compare488 struct callback_userdata {489 bool ok;490 double max_err;491 ggml_backend_t backend1;492 ggml_backend_t backend2;493 };494 495 callback_userdata ud {496 true,497 max_nmse_err(),498 backend1,499 backend2500 };501 502 auto callback = [](int index, ggml_tensor * t1, ggml_tensor * t2, void * user_data) -> bool {503 callback_userdata * ud = (callback_userdata *) user_data;504 const char * bn1 = ggml_backend_name(ud->backend1);505 const char * bn2 = ggml_backend_name(ud->backend2);506 507 if (t1->op == GGML_OP_NONE) {508 // sentinels must be unchanged509 std::vector<uint8_t> t1_data(ggml_nbytes(t1));510 std::vector<uint8_t> t2_data(ggml_nbytes(t2));511 ggml_backend_tensor_get(t1, t1_data.data(), 0, ggml_nbytes(t1));512 ggml_backend_tensor_get(t2, t2_data.data(), 0, ggml_nbytes(t2));513 514 if (memcmp(t1_data.data(), t2_data.data(), ggml_nbytes(t1)) != 0) {515 printf("sentinel mismatch: %s ", t1->name);516 ud->ok = false;517 return true;518 }519 }520 521 std::vector<float> f1 = tensor_to_float(t1);522 std::vector<float> f2 = tensor_to_float(t2);523 524 for (size_t i = 0; i < f1.size(); i++) {525 // check for nans526 if (std::isnan(f1[i]) || std::isnan(f2[i])) {527 printf("[%s] NaN at index %zu (%s=%f %s=%f) ", ggml_op_desc(t1), i, bn1, f1[i], bn2, f2[i]);528 ud->ok = false;529 return true;530 }531 // check for infs: both must be inf of the same sign, or both must be finite532 if (isinf_or_max(f1[i]) || isinf_or_max(f2[i])) {533 if (isinf_or_max(f1[i]) && isinf_or_max(f2[i])) {534 if (std::signbit(f1[i]) != std::signbit(f2[i])) {535 printf("[%s] inf sign mismatch: %s=%f %s=%f ", ggml_op_desc(t1), bn1, f1[i], bn2, f2[i]);536 ud->ok = false;537 return true;538 }539 } else {540 printf("[%s] inf mismatch: %s=%f %s=%f ", ggml_op_desc(t1), bn1, f1[i], bn2, f2[i]);541 ud->ok = false;542 return true;543 }544 }545 }546 547 double err = nmse(f1.data(), f2.data(), f1.size());548 if (err > ud->max_err) {549 printf("[%s] NMSE = %.9f > %.9f ", ggml_op_desc(t1), err, ud->max_err);550 //for (int i = 0; i < (int) f1.size(); i++) {551 // printf("%5d %9.6f %9.6f, diff = %9.6f\n", i, f1[i], f2[i], f1[i] - f2[i]);552 //}553 //printf("\n");554 //exit(1);555 ud->ok = false;556 }557 return true;558 559 GGML_UNUSED(index);560 };561 562 const bool cmp_ok = ggml_backend_compare_graph_backend(backend1, backend2, gf, callback, &ud);563 564 if (!cmp_ok) {565 printf("compare failed ");566 }567 568 ggml_backend_buffer_free(buf);569 570 ggml_free(ctx);571 572 if (ud.ok && cmp_ok) {573 printf("\033[1;32mOK\033[0m\n");574 return true;575 }576 577 printf("\033[1;31mFAIL\033[0m\n");578 return false;579 }580 581 bool eval_perf(ggml_backend_t backend, const char * op_name) {582 mode = MODE_PERF;583 584 static const size_t graph_nodes = 8192;585 586 ggml_init_params params = {587 /* .mem_size = */ ggml_tensor_overhead()*128 + ggml_graph_overhead_custom(graph_nodes, false),588 /* .mem_base = */ NULL,589 /* .no_alloc = */ true,590 };591 ggml_context * ctx = ggml_init(params);592 GGML_ASSERT(ctx);593 594 ggml_tensor * out = build_graph(ctx);595 596 if (op_name != nullptr && op_desc(out) != op_name) {597 //printf(" %s: skipping\n", op_desc(out).c_str());598 ggml_free(ctx);599 return true;600 }601 602 int len = printf(" %s(%s): ", op_desc(out).c_str(), vars().c_str());603 fflush(stdout);604 605 // check if backends support op606 if (!ggml_backend_supports_op(backend, out)) {607 printf("not supported\n");608 ggml_free(ctx);609 return true;610 }611 612 // align while also leaving some margin for variations in parameters613 int align = 8;614 int last = (len + align - 1) / align * align;615 if (last - len < 5) {616 last += align;617 }618 printf("%*s", last - len, "");619 620 // allocate621 ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors(ctx, backend);622 if (buf == NULL) {623 printf("failed to allocate tensors\n");624 ggml_free(ctx);625 return false;626 }627 628 // randomize tensors629 initialize_tensors(ctx);630 631 // build graph632 ggml_cgraph * gf = ggml_new_graph_custom(ctx, graph_nodes, false);633 ggml_build_forward_expand(gf, out);634 635 // warmup run636 ggml_backend_graph_compute(backend, gf);637 638 // determine number of runs639 int n_runs;640 bool is_cpu = ggml_backend_dev_type(ggml_backend_get_device(backend)) == GGML_BACKEND_DEVICE_TYPE_CPU;641 if (op_flops(out) > 0) {642 // based on flops643 const uint64_t GFLOP = 1000 * 1000 * 1000;644 const uint64_t target_flops_cpu = 8ULL * GFLOP;645 const uint64_t target_flops_gpu = 100ULL * GFLOP;646 uint64_t target_flops = is_cpu ? target_flops_cpu : target_flops_gpu;647 n_runs = std::min<int>(ggml_graph_size(gf) - ggml_graph_n_nodes(gf), target_flops / op_flops(out)) + 1;648 } else {649 // based on memory size650 const size_t GB = 1ULL << 30;651 const size_t target_size_cpu = 8 * GB;652 const size_t target_size_gpu = 32 * GB;653 size_t target_size = is_cpu ? target_size_cpu : target_size_gpu;654 n_runs = std::min<int>(ggml_graph_size(gf) - ggml_graph_n_nodes(gf), target_size / op_size(out)) + 1;655 }656 657 // duplicate the op658 for (int i = 1; i < n_runs; i++) {659 ggml_graph_add_node(gf, out);660 }661 662 // calculate memory663 size_t mem = n_runs * op_size(out);664 auto tensor_op_size = [](ggml_tensor * t) {665 size_t size = ggml_nbytes(t);666 // add source tensors667 for (int i = 0; i < GGML_MAX_SRC; i++) {668 if (t->src[i] != NULL) {669 size += ggml_nbytes(t->src[i]);670 }671 }672 return size;673 };674 for (int i = 0; i < ggml_graph_n_nodes(gf); ++i) {675 if (ggml_is_view_op(ggml_graph_node(gf, i)->op) || ggml_graph_node(gf, i) == out) {676 continue;677 }678 mem += tensor_op_size(ggml_graph_node(gf, i));679 }680 681 // run682 int64_t total_time_us = 0;683 int64_t total_mem = 0;684 int total_runs = 0;685 do {686 int64_t start_time = ggml_time_us();687 ggml_backend_graph_compute(backend, gf);688 int64_t end_time = ggml_time_us();689 690 total_time_us += end_time - start_time;691 total_mem += mem;692 total_runs += n_runs;693 } while (total_time_us < 1000*1000); // run for at least 1 second694 695 printf(" %8d runs - %8.2f us/run - ",696 total_runs,697 (double)total_time_us / total_runs);698 699 if (op_flops(out) > 0) {700 double flops_per_sec = (op_flops(out) * total_runs) / (total_time_us / 1e6);701 auto format_flops = [](double flops) -> std::string {702 char buf[256];703 if (flops >= 1e12) {704 snprintf(buf, sizeof(buf), "%6.2f TFLOP", flops / 1e12);705 } else if (flops >= 1e9) {706 snprintf(buf, sizeof(buf), "%6.2f GFLOP", flops / 1e9);707 } else if (flops >= 1e6) {708 snprintf(buf, sizeof(buf), "%6.2f MFLOP", flops / 1e6);709 } else {710 snprintf(buf, sizeof(buf), "%6.2f KFLOP", flops / 1e3);711 }712 return buf;713 };714 printf("%s/run - \033[1;34m%sS\033[0m",715 format_flops(op_flops(out)).c_str(),716 format_flops(flops_per_sec).c_str());717 718 } else {719 printf("%8zu kB/run - \033[1;34m%7.2f GB/s\033[0m",720 op_size(out) / 1024,721 total_mem / (total_time_us / 1e6) / 1024.0 / 1024.0 / 1024.0);722 }723 printf("\n");724 725 ggml_backend_buffer_free(buf);726 727 ggml_free(ctx);728 729 return true;730 }731 732 bool eval_grad(ggml_backend_t backend, const char * op_name) {733 mode = MODE_GRAD;734 const std::vector<float> expect = grad_expect();735 736 ggml_init_params params = {737 /* .mem_size = */ ggml_tensor_overhead()*128 + 2*ggml_graph_overhead_custom(GGML_DEFAULT_GRAPH_SIZE, true),738 /* .mem_base = */ NULL,739 /* .no_alloc = */ true,740 };741 ggml_context * ctx = ggml_init(params);742 GGML_ASSERT(ctx);743 744 gf = ggml_new_graph_custom(ctx, GGML_DEFAULT_GRAPH_SIZE, true);745 gb = ggml_new_graph_custom(ctx, GGML_DEFAULT_GRAPH_SIZE, true);746 747 ggml_tensor * out = build_graph(ctx);748 749 if ((op_name != nullptr && op_desc(out) != op_name) || out->op == GGML_OP_OPT_STEP_ADAMW) {750 //printf(" %s: skipping\n", op_desc(out).c_str());751 ggml_free(ctx);752 return true;753 }754 755 printf(" %s(%s): ", op_desc(out).c_str(), vars().c_str());756 fflush(stdout);757 758 if (out->type != GGML_TYPE_F32) {759 ggml_free(ctx);760 printf("not supported [%s->type != FP32]\n", out->name);761 return true;762 }763 764 // check if the backend supports the ops765 bool supported = true;766 bool any_params = false;767 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {768 if (!ggml_backend_supports_op(backend, t)) {769 printf("not supported [%s] ", ggml_backend_name(backend));770 supported = false;771 break;772 }773 if ((t->flags & GGML_TENSOR_FLAG_PARAM)) {774 any_params = true;775 if (t->type != GGML_TYPE_F32) {776 printf("not supported [%s->type != FP32] ", t->name);777 supported = false;778 break;779 }780 }781 }782 if (!any_params) {783 printf("not supported [%s] \n", op_desc(out).c_str());784 supported = false;785 }786 if (!supported) {787 printf("\n");788 ggml_free(ctx);789 return true;790 }791 792 int64_t ngrads = 0;793 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {794 if (t->flags & GGML_TENSOR_FLAG_PARAM) {795 ngrads += ggml_nelements(t);796 }797 }798 if (ngrads > grad_nmax()) {799 printf("skipping large tensors for speed \n");800 ggml_free(ctx);801 return true;802 }803 804 805 if (!ggml_is_scalar(out)) {806 out = ggml_sum(ctx, out);807 ggml_set_name(out, "sum_of_out");808 }809 ggml_set_loss(out);810 811 ggml_build_forward_expand(gf, out);812 ggml_graph_cpy(gf, gb);813 ggml_build_backward_expand(ctx, ctx, gb, false);814 if (expect.size() != 1 || expect[0] != 0.0f) {815 GGML_ASSERT(ggml_graph_n_nodes(gb) > ggml_graph_n_nodes(gf));816 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {817 GGML_ASSERT(!(t->flags & GGML_TENSOR_FLAG_PARAM) || ggml_graph_get_grad(gb, t)->op != GGML_OP_NONE);818 }819 }820 821 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {822 if (!ggml_backend_supports_op(backend, t)) {823 printf("not supported [%s] ", ggml_backend_name(backend));824 supported = false;825 break;826 }827 if ((t->flags & GGML_TENSOR_FLAG_PARAM) && t->type != GGML_TYPE_F32) {828 printf("not supported [%s->type != FP32] ", t->name);829 supported = false;830 break;831 }832 }833 if (!supported) {834 printf("\n");835 ggml_free(ctx);836 return true;837 }838 839 // allocate840 ggml_backend_buffer_t buf = ggml_backend_alloc_ctx_tensors(ctx, backend);841 if (buf == NULL) {842 printf("failed to allocate tensors [%s] ", ggml_backend_name(backend));843 ggml_free(ctx);844 return false;845 }846 847 848 initialize_tensors(ctx); // Randomizes all tensors (including gradients).849 ggml_graph_reset(gb); // Sets gradients to 1 if loss, 0 otherwise.850 851 ggml_backend_graph_compute(backend, gf);852 ggml_backend_graph_compute(backend, gb);853 854 bool ok = true;855 for (struct ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {856 if (!(t->flags & GGML_TENSOR_FLAG_PARAM)) {857 continue;858 }859 860 const char * bn = ggml_backend_name(backend);861 const int64_t ne = ggml_nelements(t);862 863 std::vector<float> ga;864 struct ggml_tensor * grad = ggml_graph_get_grad(gb, t);865 if (grad) {866 ga = tensor_to_float(grad);867 } else {868 ga.resize(ne); // default value is 0.0f869 }870 871 for (int64_t i = 0; i < ne; ++i) { // gradient algebraic872 // check for nans873 if (!std::isfinite(ga[i])) {874 printf("[%s] nonfinite gradient at index %" PRId64 " (%s=%f) ", ggml_op_desc(t), i, bn, ga[i]);875 ok = false;876 break;877 }878 }879 if (!ok) {880 break;881 }882 883 std::vector<float> gn(ne); // gradient numeric884 GGML_ASSERT(ga.size() == gn.size());885 886 std::vector<float> x0 = tensor_to_float(t); // original t data887 GGML_ASSERT(ggml_is_scalar(out));888 GGML_ASSERT(out->type == GGML_TYPE_F32);889 890 const float eps = grad_eps();891 for (int64_t i = 0; i < ne; ++i) {892 const float xiu = x0[i] + 1.0f*eps; // x, index i, up893 const float xiuh = x0[i] + 0.5f*eps; // x, index i, up half894 const float xidh = x0[i] - 0.5f*eps; // x, index i, down half895 const float xid = x0[i] - 1.0f*eps; // x, index i, down896 897 float fu, fuh, fdh, fd; // output values for xiu, xiuh, xid, xidh898 899 ggml_backend_tensor_set(t, &xiu, i*sizeof(float), sizeof(float));900 ggml_backend_graph_compute(backend, gf);901 ggml_backend_tensor_get(out, &fu, 0, ggml_nbytes(out));902 903 ggml_backend_tensor_set(t, &xid, i*sizeof(float), sizeof(float));904 ggml_backend_graph_compute(backend, gf);905 ggml_backend_tensor_get(out, &fd, 0, ggml_nbytes(out));906 907 if (grad_precise()) {908 ggml_backend_tensor_set(t, &xiuh, i*sizeof(float), sizeof(float));909 ggml_backend_graph_compute(backend, gf);910 ggml_backend_tensor_get(out, &fuh, 0, ggml_nbytes(out));911 912 ggml_backend_tensor_set(t, &xidh, i*sizeof(float), sizeof(float));913 ggml_backend_graph_compute(backend, gf);914 ggml_backend_tensor_get(out, &fdh, 0, ggml_nbytes(out));915 916 gn[i] = (8.0*(double)fuh + (double)fd - (8.0*(double)fdh + (double)fu)) / (6.0*(double)eps);917 } else {918 gn[i] = (fu - fd) / (2.0f*eps);919 }920 921 ggml_backend_tensor_set(t, x0.data(), 0, ggml_nbytes(t));922 }923 924 const double err = mean_abs_asymm(gn.data(), ga.data(), gn.size(), expect);925 if (err > max_maa_err()) {926 printf("[%s] MAA = %.9f > %.9f ", ggml_op_desc(t), err, max_maa_err());927 ok = false;928 break;929 }930 if (!ok) {931 break;932 }933 }934 935 if (!ok) {936 printf("compare failed ");937 }938 939 ggml_backend_buffer_free(buf);940 941 ggml_free(ctx);942 943 if (ok) {944 printf("\033[1;32mOK\033[0m\n");945 return true;946 }947 948 printf("\033[1;31mFAIL\033[0m\n");949 return false;950 }951};952 953 954// ###################################955// ## Section 2: GGML Op Defintions ##956// ###################################957 958 959// The following is an example showing the bare minimum for creating a test for a GGML op.960 961// GGML_OP_EXAMPLE962struct test_example : public test_case {963 // Always define these 2 or variants thereof:964 const ggml_type type; // The type of the input tensors.965 const std::array<int64_t, 4> ne; // The shape of the input tensors.966 // For some ops it's necessary to define multiple types or shapes for the inputs.967 // Or they may need additional parameters.968 969 // Put all parameters needed to fully define the test into one of the VARS_TO_STR macros.970 // In most cases these are just the properties of the struct that you defined above.971 // This is needed for info prints.972 std::string vars() override {973 return VARS_TO_STR2(type, ne);974 }975 976 // Define a constructor for the struct.977 // In most cases it will be sufficient to have the same arguments as the struct has properties978 // and just use initializer lists.979 test_example(ggml_type type = GGML_TYPE_F32,980 std::array<int64_t, 4> ne = {10, 5, 4, 3})981 : type(type), ne(ne) {}982 983 // Define how a simple GGML compute graph can be constructed for the new GGML op.984 ggml_tensor * build_graph(ggml_context * ctx) override {985 // Step 1: create input tensors that don't depend on any other tensors:986 ggml_tensor * a = ggml_new_tensor(ctx, type, 4, ne.data());987 ggml_set_name(a, "a"); // Setting names is optional but it's useful for debugging.988 989 ggml_tensor * b = ggml_new_tensor(ctx, type, 4, ne.data());990 ggml_set_name(b, "b");991 992 // Step 2: use the op that you want to test in the GGML compute graph.993 ggml_tensor * out = ggml_add(ctx, a, b); // For this example we're just doing a simple addition.994 ggml_set_name(out, "out");995 996 // Step 3: return the output tensor.997 return out;998 }999 // In order to also check the gradients for your op, add calls like ggml_set_param(ctx, a)1000 // immediately after you create the tensors.1001 // This is optional and only makes sense if a backward pass has actually been implemented for the new op.1002};1003 1004 1005// GGML_OP_UNARY1006struct test_unary : public test_case {1007 const ggml_unary_op op;1008 const ggml_type type;1009 const std::array<int64_t, 4> ne_a;1010 int v; // view (1 : non-contiguous a)1011 1012 std::string vars() override {1013 return VARS_TO_STR3(type, ne_a, v);1014 }1015 1016 test_unary(ggml_unary_op op,1017 ggml_type type = GGML_TYPE_F32,1018 std::array<int64_t, 4> ne_a = {128, 2, 2, 2},1019 int v = 0)1020 : op(op), type(type), ne_a(ne_a), v(v) {}1021 1022 ggml_tensor * build_graph(ggml_context * ctx) override {1023 const bool grad_supported = op == GGML_UNARY_OP_ABS || op == GGML_UNARY_OP_SGN || op == GGML_UNARY_OP_NEG ||1024 op == GGML_UNARY_OP_STEP || op == GGML_UNARY_OP_RELU || op == GGML_UNARY_OP_SILU;1025 1026 ggml_tensor * a;1027 if (v & 1) {1028 auto ne = ne_a; ne[0] *= 3;1029 a = ggml_new_tensor(ctx, type, 4, ne.data());1030 if (grad_supported) {1031 ggml_set_param(ctx, a);1032 }1033 ggml_set_name(a, "a");1034 1035 a = ggml_view_4d(ctx, a, ne_a[0], ne_a[1], ne_a[2], ne_a[3], a->nb[1], a->nb[2], a->nb[3], 0);1036 ggml_set_name(a, "view_of_a");1037 } else {1038 a = ggml_new_tensor(ctx, type, 4, ne_a.data());1039 if (grad_supported) {1040 ggml_set_param(ctx, a);1041 }1042 ggml_set_name(a, "a");1043 }1044 1045 ggml_tensor * out = ggml_unary(ctx, a, op);1046 ggml_set_name(out, "out");1047 1048 return out;1049 }1050 1051 void initialize_tensors(ggml_context * ctx) override {1052 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {1053 // test extended range of values to check for NaNs in GELU1054 init_tensor_uniform(t, -150.f, 150.f);1055 }1056 }1057 1058 float grad_eps() override {1059 return 15.0f;1060 }1061 1062 std::vector<float> grad_expect() override {1063 if (op == GGML_UNARY_OP_ABS) {1064 return {-1.0f, 1.0f};1065 }1066 if (op == GGML_UNARY_OP_SGN || op == GGML_UNARY_OP_STEP) {1067 return {0.0f};1068 }1069 if (op == GGML_UNARY_OP_RELU) {1070 return {0.0f, 1.0f};1071 }1072 return {};1073 }1074 1075};1076 1077// GGML_OP_GET_ROWS1078struct test_get_rows : public test_case {1079 const ggml_type type;1080 const int n; // cols1081 const int m; // rows1082 const int r; // rows to get1083 const int b; // batch size1084 const bool v; // view (non-contiguous src1)1085 1086 std::string vars() override {1087 return VARS_TO_STR6(type, n, m, r, b, v);1088 }1089 1090 test_get_rows(ggml_type type = GGML_TYPE_F32, int n = 10, int m = 5, int r = 3, int b = 1, bool v = false)1091 : type(type), n(n), m(m), r(r), b(b), v(v) {}1092 1093 ggml_tensor * build_graph(ggml_context * ctx) override {1094 ggml_tensor * in = ggml_new_tensor_3d(ctx, type, n, m, b);1095 ggml_set_name(in, "in");1096 1097 ggml_tensor * rows = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, r, b);1098 ggml_set_name(rows, "rows");1099 if (v) {1100 rows = ggml_view_2d(ctx, rows, r/2, b, rows->nb[1], 0);1101 ggml_set_name(rows, "view_of_rows");1102 }1103 1104 const bool grad_supported = ggml_is_matrix(in) && ggml_is_vector(rows);1105 if (grad_supported) {1106 ggml_set_param(ctx, in);1107 // rows is a constant input -> no gradients1108 }1109 1110 ggml_tensor * out = ggml_get_rows(ctx, in, rows);1111 ggml_set_name(out, "out");1112 1113 return out;1114 }1115 1116 void initialize_tensors(ggml_context * ctx) override {1117 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {1118 if (t->type == GGML_TYPE_I32) {1119 if (ggml_is_view_op(t->op)) { continue; }1120 // rows1121 std::vector<int> data(r*b);1122 for (int i = 0; i < r*b; i++) {1123 data[i] = rand() % m;1124 }1125 ggml_backend_tensor_set(t, data.data(), 0, r * b * sizeof(int));1126 } else {1127 init_tensor_uniform(t);1128 }1129 }1130 }1131};1132 1133// GGML_OP_GET_ROWS_BACK1134struct test_get_rows_back : public test_case {1135 const ggml_type type;1136 const int n; // cols1137 const int m; // rows1138 const int r; // rows to get1139 const int b; // batch size1140 const bool v; // view (non-contiguous src1)1141 1142 std::string vars() override {1143 return VARS_TO_STR6(type, n, m, r, b, v);1144 }1145 1146 test_get_rows_back(ggml_type type = GGML_TYPE_F32, int n = 10, int m = 5, int r = 3, int b = 1, bool v = false)1147 : type(type), n(n), m(m), r(r), b(b), v(v) {}1148 1149 ggml_tensor * build_graph(ggml_context * ctx) override {1150 ggml_tensor * in_forward = ggml_new_tensor_3d(ctx, type, n, m, b);1151 ggml_set_name(in_forward, "in_forward");1152 1153 ggml_tensor * rows = ggml_new_tensor_2d(ctx, GGML_TYPE_I32, r, b);1154 ggml_set_name(rows, "rows");1155 if (v) {1156 rows = ggml_view_2d(ctx, rows, r/2, b, rows->nb[1], 0);1157 ggml_set_name(rows, "view_of_rows");1158 }1159 1160 ggml_tensor * grad = ggml_new_tensor_3d(ctx, type, n, r, b);1161 ggml_set_name(grad, "grad");1162 1163 ggml_tensor * out = ggml_get_rows_back(ctx, grad, rows, in_forward);1164 ggml_set_name(out, "out");1165 1166 return out;1167 }1168 1169 void initialize_tensors(ggml_context * ctx) override {1170 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != NULL; t = ggml_get_next_tensor(ctx, t)) {1171 if (t->type == GGML_TYPE_I32) {1172 if (ggml_is_view_op(t->op)) { continue; }1173 // rows1174 std::vector<int> data(r*b);1175 for (int i = 0; i < r*b; i++) {1176 data[i] = rand() % m;1177 }1178 ggml_backend_tensor_set(t, data.data(), 0, r * b * sizeof(int));1179 } else {1180 init_tensor_uniform(t);1181 }1182 }1183 }1184};1185 1186// GGML_OP_ARGMAX1187struct test_argmax : public test_case {1188 const ggml_type type;1189 const std::array<int64_t, 4> ne;1190 1191 std::string vars() override {1192 return VARS_TO_STR2(type, ne);1193 }1194 1195 test_argmax(ggml_type type = GGML_TYPE_F32,1196 std::array<int64_t, 4> ne = {10, 100, 1, 1})1197 : type(type), ne(ne) {}1198 1199 ggml_tensor * build_graph(ggml_context * ctx) override {1200 ggml_tensor * a = ggml_new_tensor(ctx, type, 4, ne.data());