Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1// This file defines tests for various GGML ops and backends.2// For the forward pass it asserts that the results of multiple backends computing the same GGML ops are consistent.3// For the backward pass it asserts that the gradients from backpropagation are consistent4// with the gradients obtained via the method of finite differences ("grad" mode, this is optional).5// It is also possible to check the performance ("perf" mode).6//7// this file has three sections: Section 1 does general setup, section 2 defines the GGML ops to be tested,8// and section 3 defines which tests to run.9// Quick start for adding a new GGML op: Go to section 2 and create a struct that inherits from test_case,10// then go to section 3 and add an instantiation of your struct.11 12 13// ##############################14// ## Section 1: General Setup ##15// ##############################16 17 18#include "ggml.h"19#include "ggml-alloc.h"20#include "ggml-backend.h"21#include "ggml-cpp.h"22 23#include <algorithm>24#include <atomic>25#include <array>26#include <cfloat>27#include <cinttypes>28#include <cstdarg>29#include <cstdint>30#include <cstdio>31#include <cstdlib>32#include <cstring>33#include <ctime>34#include <future>35#include <fstream>36#include <memory>37#include <mutex>38#include <random>39#include <regex>40#include <set>41#include <sstream>42#include <string>43#include <string_view>44#include <thread>45#include <vector>46#include <unordered_map>47 48#ifdef __EMSCRIPTEN__49# define N_THREADS 150#else51# define N_THREADS std::thread::hardware_concurrency()52#endif53 54static void init_tensor_uniform(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) {55 size_t nels = ggml_nelements(tensor);56 std::vector<float> data(nels);57 {58 // parallel initialization59 static const size_t n_threads = N_THREADS;60 61 auto init_thread = [&](size_t start, size_t end) {62 thread_local std::default_random_engine gen(std::random_device{}());63 std::uniform_real_distribution<float> distribution(min, max);64 for (size_t i = start; i < end; i++) {65 data[i] = distribution(gen);66 }67 };68 69 if (n_threads == 1) {70 init_thread(0, nels);71 } else {72 std::vector<std::future<void>> tasks;73 tasks.reserve(n_threads);74 for (size_t i = 0; i < n_threads; i++) {75 size_t start = i*nels/n_threads;76 size_t end = (i+1)*nels/n_threads;77 tasks.push_back(std::async(std::launch::async, init_thread, start, end));78 }79 for (auto & t : tasks) {80 t.get();81 }82 }83 }84 85 if (tensor->type == GGML_TYPE_F32 || tensor->type == GGML_TYPE_I32) {86 ggml_backend_tensor_set(tensor, data.data(), 0, nels * sizeof(float));87 } else if (ggml_is_quantized(tensor->type) || tensor->type == GGML_TYPE_F16 || tensor->type == GGML_TYPE_BF16) {88 GGML_ASSERT(nels % ggml_blck_size(tensor->type) == 0);89 90 // dummy importance matrix91 std::vector<float> imatrix(tensor->ne[0], 1.0f);92 const float * im = imatrix.data();93 if (!ggml_quantize_requires_imatrix(tensor->type)) {94 // when the imatrix is optional, we want to test both quantization with and without imatrix95 // use one of the random numbers to decide96 if (data[0] > 0.5f*(min + max)) {97 im = nullptr;98 }99 }100 101 std::vector<uint8_t> dataq(ggml_row_size(tensor->type, nels));102 {103 // parallel quantization by block104 size_t blck_size = ggml_blck_size(tensor->type);105 size_t n_blocks = nels / blck_size;106 107 auto quantize_thread = [&](size_t start, size_t end) {108 ggml_quantize_chunk(tensor->type, data.data(), dataq.data(),109 start * blck_size, end - start, blck_size, im);110 };111 112 const size_t min_blocks_per_thread = 1;113 const size_t n_quant_threads = std::min<size_t>(std::max<size_t>(N_THREADS/2, 1),114 std::max<size_t>(1, n_blocks / min_blocks_per_thread));115 116 if (n_quant_threads == 1) {117 // single-threaded quantization: do all blocks in the current thread118 quantize_thread(0, n_blocks);119 } else {120 std::vector<std::future<void>> tasks;121 tasks.reserve(n_quant_threads);122 for (size_t i = 0; i < n_quant_threads; i++) {123 size_t start = i*n_blocks/n_quant_threads;124 size_t end = (i+1)*n_blocks/n_quant_threads;125 tasks.push_back(std::async(std::launch::async, quantize_thread, start, end));126 }127 for (auto & t : tasks) {128 t.get();129 }130 }131 }132 ggml_backend_tensor_set(tensor, dataq.data(), 0, dataq.size());133 } else if (tensor->type == GGML_TYPE_I8 || tensor->type == GGML_TYPE_I16) {134 // This is going to create some weird integers though.135 ggml_backend_tensor_set(tensor, data.data(), 0, nels * ggml_type_size(tensor->type));136 } else if (tensor->type == GGML_TYPE_I64) {137 // Integers with a size of 8 bytes can be set by mirroring the float data, the specific values are again not really meaningful.138 const size_t nbytes_half = nels * sizeof(float);139 ggml_backend_tensor_set(tensor, data.data(), 0*nbytes_half, nbytes_half);140 ggml_backend_tensor_set(tensor, data.data(), 1*nbytes_half, nbytes_half);141 } else {142 GGML_ABORT("fatal error");143 }144}145 146// generate an F16 mask where certain blocks are randomly masked with -INF value147static void init_tensor_kq_mask(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) {148 GGML_ASSERT(tensor->type == GGML_TYPE_F16);149 150 GGML_TENSOR_LOCALS( int32_t, ne, tensor, ne);151 152 std::vector<float> data_f32(ne0*ne1*ne2*ne3);153 std::vector<ggml_fp16_t> data_f16(ne0*ne1*ne2*ne3);154 155 std::random_device rd;156 std::mt19937 gen(rd());157 std::uniform_real_distribution<float> dis(min, max);158 159 for (size_t i = 0; i < data_f32.size(); i++) {160 data_f32[i] = dis(gen);161 }162 163 // block size164 const int blck0 = 128;165 const int blck1 = 64;166 167 // number of INF/zero blocks168 const int n_inf_zero_blocks = 0.2*(ne0*ne1*ne2*ne3)/(blck0*blck1);169 170 for (int b = 0; b < n_inf_zero_blocks; b++) {171 const int p3 = (rd() % ne3);172 const int p2 = (rd() % ne2);173 const int p1 = (rd() % ne1);174 const int p0 = (rd() % ne0);175 176 bool inf = rd() & 1;177 178 for (int i1 = 0; i1 < blck1 && p1 + i1 < ne1; i1++) {179 const int idx = p3*ne2*ne1*ne0 + p2*ne1*ne0 + (p1 + i1)*ne0 + p0;180 181 for (int i0 = 0; i0 < blck0 && p0 + i0 < ne0; i0++) {182 data_f32[idx + i0] = inf ? -INFINITY : 0.0f;183 }184 }185 }186 187 ggml_fp32_to_fp16_row(data_f32.data(), data_f16.data(), ne0*ne1*ne2*ne3);188 189 ggml_backend_tensor_set(tensor, data_f16.data(), 0, data_f16.size()*sizeof(ggml_fp16_t));190}191 192// generate a lower triangular matrix193static void init_tensor_tril(ggml_tensor * tensor, float min = -1.0f, float max = 1.0f) {194 GGML_ASSERT(tensor->type == GGML_TYPE_F32);195 GGML_ASSERT(tensor->ne[0] == tensor->ne[1]);196 197 GGML_TENSOR_LOCALS(int32_t, ne, tensor, ne);198 GGML_TENSOR_LOCALS(size_t, nb, tensor, nb);199 200 std::vector<float> data_f32(ne0*ne1*ne2*ne3);201 202 std::random_device rd;203 std::mt19937 gen(rd());204 std::uniform_real_distribution<float> dis(min, max);205 206 for (int64_t i3 = 0; i3 < ne3; i3++) {207 for (int64_t i2 = 0; i2 < ne2; i2++) {208 for (int64_t i1 = 0; i1 < ne1; i1++) {209 for (int64_t i0 = 0; i0 < ne0; i0++) {210 int64_t idx = (i0 * nb0 + i1 * nb1 + i2 * nb2 + i3 * nb3) / sizeof(float);211 if (i0 <= i1) {212 data_f32[idx] = dis(gen);213 } else {214 data_f32[idx] = 0.0f;215 }216 }217 }218 }219 }220 221 ggml_backend_tensor_set(tensor, data_f32.data(), 0, ggml_nbytes(tensor));222}223 224static std::vector<float> tensor_to_float(const ggml_tensor * t) {225 std::vector<float> tv;226 tv.reserve(ggml_nelements(t));227 228 std::vector<uint8_t> buf(ggml_nbytes(t));229 ggml_backend_tensor_get(t, buf.data(), 0, ggml_nbytes(t));230 231 const auto * tt = ggml_get_type_traits(t->type);232 size_t bs = ggml_blck_size(t->type);233 std::vector<float> vq(ggml_blck_size(t->type));234 bool quantized = ggml_is_quantized(t->type);235 236 // access elements by index to avoid gaps in views237 for (int64_t i3 = 0; i3 < t->ne[3]; i3++) {238 for (int64_t i2 = 0; i2 < t->ne[2]; i2++) {239 for (int64_t i1 = 0; i1 < t->ne[1]; i1++) {240 for (int64_t i0 = 0; i0 < t->ne[0]; i0 += bs) {241 size_t i = i3*t->nb[3] + i2*t->nb[2] + i1*t->nb[1] + i0/bs*t->nb[0];242 if (t->type == GGML_TYPE_F16) {243 tv.push_back(ggml_fp16_to_fp32(*(ggml_fp16_t*)&buf[i]));244 } else if (t->type == GGML_TYPE_BF16) {245 tv.push_back(ggml_bf16_to_fp32(*(ggml_bf16_t*)&buf[i]));246 } else if (t->type == GGML_TYPE_F32) {247 tv.push_back(*(float *) &buf[i]);248 } else if (t->type == GGML_TYPE_I64) {249 tv.push_back((float)*(int64_t *) &buf[i]);250 } else if (t->type == GGML_TYPE_I32) {251 tv.push_back((float)*(int32_t *) &buf[i]);252 } else if (t->type == GGML_TYPE_I16) {253 tv.push_back((float)*(int16_t *) &buf[i]);254 } else if (t->type == GGML_TYPE_I8) {255 tv.push_back((float)*(int8_t *) &buf[i]);256 } else if (quantized) {257 tt->to_float(&buf[i], vq.data(), bs);258 tv.insert(tv.end(), vq.begin(), vq.end());259 } else {260 GGML_ABORT("fatal error");261 }262 }263 }264 }265 }266 267 return tv;268}269 270// normalized mean squared error = mse(a, b) / mse(a, 0)271static double nmse(const float * a, const float * b, size_t n) {272 double mse_a_b = 0.0;273 double mse_a_0 = 0.0;274 275 for (size_t i = 0; i < n; i++) {276 float a_i = a[i];277 float b_i = b[i];278 279 mse_a_b += (a_i - b_i) * (a_i - b_i);280 mse_a_0 += a_i * a_i;281 }282 283 return mse_a_b / mse_a_0;284}285 286// difference between 2 sets (Jaccard distance, 0 - no difference, 1 - no overlap)287template <typename T>288static double jdst(const T * a, const T * b, size_t n) {289 std::unordered_map<T, size_t> set_a;290 std::unordered_map<T, size_t> set_b;291 292 for (size_t i = 0; i < n; ++i) {293 set_a[a[i]]++;294 set_b[b[i]]++;295 }296 297 size_t diff = 0;298 299 for (const auto & p : set_a) {300 const int64_t na = p.second;301 const int64_t nb = set_b.find(p.first) != set_b.end() ? set_b.at(p.first) : 0;302 303 diff += std::abs(na - nb);304 }305 306 for (const auto & p : set_b) {307 if (set_a.find(p.first) == set_a.end()) {308 diff += p.second;309 }310 }311 312 return (double) diff / (2*n);313}314 315// maximum absolute asymmetry between a and b316// asymmetry: (a - b) / (a + b)317// This is more stable than relative error if one of the values fluctuates towards zero.318// n: number of values to compare.319// expected_vals: optional vector of expected values for a. If expected_vals is not empty, filter out all comparisons where320// a does not match any of the expected values. Needed for noncontinuous gradients where the numerical calculation can fail.321static double mean_abs_asymm(const float * a, const float * b, const size_t n, const std::vector<float> & expected_vals) {322 double sum = 0.0f;323 324 size_t nvalid = 0;325 for (size_t i = 0; i < n; i++) {326 if (!expected_vals.empty()) {327 bool matches_any = false;328 for (const float & ev : expected_vals) {329 if (fabsf(a[i] - ev) < 1e-3f) {330 matches_any = true;331 break;332 }333 }334 if (!matches_any) {335 continue;336 }337 }338 339 const float asymm = (a[i] - b[i]) / (a[i] + b[i]);340 341 sum += fabsf(asymm);342 nvalid++;343 }344 345 return sum/nvalid;346}347 348// utils for printing the variables of the test cases349 350static std::string var_to_str(const std::string & x) {351 return x;352}353 354template<typename T>355static std::string var_to_str(const T & x) {356 return std::to_string(x);357}358 359template<typename T, size_t N>360static std::string var_to_str(const T (&x)[N]) {361 std::string s = "[";362 for (size_t i = 0; i < N; i++) {363 if (i > 0) {364 s += ",";365 }366 s += var_to_str(x[i]);367 }368 s += "]";369 return s;370}371 372template<typename T, size_t N>373static std::string var_to_str(const std::array<T, N> & x) {374 std::string s = "[";375 for (size_t i = 0; i < N; i++) {376 if (i > 0) {377 s += ",";378 }379 s += var_to_str(x[i]);380 }381 s += "]";382 return s;383}384 385static std::string var_to_str(ggml_type type) {386 return ggml_type_name(type);387}388 389static std::string var_to_str(ggml_prec prec) {390 return prec == GGML_PREC_F32 ? "f32" : "def";391}392 393static std::string var_to_str(ggml_op_pool pool) {394 switch (pool) {395 case GGML_OP_POOL_AVG: return "avg";396 case GGML_OP_POOL_MAX: return "max";397 default: return std::to_string(pool);398 }399}400 401static std::string var_to_str(ggml_scale_mode mode) {402 std::string str;403 switch (mode & 0xFF) {404 case GGML_SCALE_MODE_NEAREST: str = "nearest"; break;405 case GGML_SCALE_MODE_BILINEAR: str = "bilinear"; break;406 case GGML_SCALE_MODE_BICUBIC: str = "bicubic"; break;407 default: str = std::to_string(mode); break;408 }409 if (mode & GGML_SCALE_FLAG_ALIGN_CORNERS) {410 str += "|align_corners";411 }412 if (mode & GGML_SCALE_FLAG_ANTIALIAS) {413 str += "|antialias";414 }415 return str;416}417 418#define VAR_TO_STR(x) (#x "=" + var_to_str(x))419 420#define VARS_TO_STR1(a) VAR_TO_STR(a)421#define VARS_TO_STR2(a, b) VAR_TO_STR(a) + "," + VAR_TO_STR(b)422#define VARS_TO_STR3(a, b, c) VAR_TO_STR(a) + "," + VARS_TO_STR2(b, c)423#define VARS_TO_STR4(a, b, c, d) VAR_TO_STR(a) + "," + VARS_TO_STR3(b, c, d)424#define VARS_TO_STR5(a, b, c, d, e) VAR_TO_STR(a) + "," + VARS_TO_STR4(b, c, d, e)425#define VARS_TO_STR6(a, b, c, d, e, f) VAR_TO_STR(a) + "," + VARS_TO_STR5(b, c, d, e, f)426#define VARS_TO_STR7(a, b, c, d, e, f, g) VAR_TO_STR(a) + "," + VARS_TO_STR6(b, c, d, e, f, g)427#define VARS_TO_STR8(a, b, c, d, e, f, g, h) VAR_TO_STR(a) + "," + VARS_TO_STR7(b, c, d, e, f, g, h)428#define VARS_TO_STR9(a, b, c, d, e, f, g, h, i) VAR_TO_STR(a) + "," + VARS_TO_STR8(b, c, d, e, f, g, h, i)429#define VARS_TO_STR10(a, b, c, d, e, f, g, h, i, j) VAR_TO_STR(a) + "," + VARS_TO_STR9(b, c, d, e, f, g, h, i, j)430#define VARS_TO_STR11(a, b, c, d, e, f, g, h, i, j, k) VAR_TO_STR(a) + "," + VARS_TO_STR10(b, c, d, e, f, g, h, i, j, k)431#define VARS_TO_STR12(a, b, c, d, e, f, g, h, i, j, k, l) VAR_TO_STR(a) + "," + VARS_TO_STR11(b, c, d, e, f, g, h, i, j, k, l)432#define VARS_TO_STR13(a, b, c, d, e, f, g, h, i, j, k, l, m) VAR_TO_STR(a) + "," + VARS_TO_STR12(b, c, d, e, f, g, h, i, j, k, l, m)433#define VARS_TO_STR14(a, b, c, d, e, f, g, h, i, j, k, l, m, n) VAR_TO_STR(a) + "," + VARS_TO_STR13(b, c, d, e, f, g, h, i, j, k, l, m, n)434#define VARS_TO_STR15(a, b, c, d, e, f, g, h, i, j, k, l, m, n, o) VAR_TO_STR(a) + "," + VARS_TO_STR14(b, c, d, e, f, g, h, i, j, k, l, m, n, o)435#define VARS_TO_STR16(a, b, c, d, e, f, g, h, i, j, k, l, m, n, o, p) VAR_TO_STR(a) + "," + VARS_TO_STR15(b, c, d, e, f, g, h, i, j, k, l, m, n, o, p)436 437#ifdef GGML_USE_SYCL438static bool inline _isinf(float f) {439 return (*(uint32_t *)&f & 0x7fffffff) == 0x7f800000;440}441#else442static bool inline _isinf(float f) { return std::isinf(f); }443#endif444 445// accept FLT_MAX as infinity446static bool isinf_or_max(float f) {447 return _isinf(f) || f == FLT_MAX || f == -FLT_MAX;448}449 450static bool ggml_is_view_op(enum ggml_op op) {451 return op == GGML_OP_VIEW || op == GGML_OP_RESHAPE || op == GGML_OP_PERMUTE || op == GGML_OP_TRANSPOSE;452}453 454static bool backend_has_feature(ggml_backend_t backend, const char * feature_name) {455 ggml_backend_dev_t dev = ggml_backend_get_device(backend);456 ggml_backend_reg_t reg = ggml_backend_dev_backend_reg(dev);457 458 auto get_features = (ggml_backend_get_features_t) ggml_backend_reg_get_proc_address(reg, "ggml_backend_get_features");459 if (!get_features) {460 return false;461 }462 463 const ggml_backend_feature * features = get_features(reg);464 if (!features) {465 return false;466 }467 468 for (const ggml_backend_feature * f = features; f->name; ++f) {469 if (strcmp(f->name, feature_name) == 0 && strcmp(f->value, "1") == 0) {470 return true;471 }472 }473 return false;474}475 476enum test_mode {477 MODE_TEST,478 MODE_PERF,479 MODE_GRAD,480 MODE_SUPPORT,481};482 483// Output format support similar to llama-bench484enum output_formats { CONSOLE, SQL, CSV };485 486static const char * output_format_str(output_formats format) {487 switch (format) {488 case CONSOLE:489 return "console";490 case SQL:491 return "sql";492 case CSV:493 return "csv";494 default:495 GGML_ABORT("invalid output format");496 }497}498 499static bool output_format_from_str(const std::string & s, output_formats & format) {500 if (s == "console") {501 format = CONSOLE;502 } else if (s == "sql") {503 format = SQL;504 } else if (s == "csv") {505 format = CSV;506 } else {507 return false;508 }509 return true;510}511 512static std::string test_time_now() {513 time_t t = time(NULL);514 struct tm tm_buf;515#ifdef _WIN32516 if (gmtime_s(&tm_buf, &t) != 0) {517 return "";518 }519#else520 if (gmtime_r(&t, &tm_buf) == nullptr) {521 return "";522 }523#endif524 char buf[32];525 if (std::strftime(buf, sizeof(buf), "%FT%TZ", &tm_buf) == 0) {526 return "";527 }528 return buf;529}530 531// Test result structure for SQL output532struct test_result {533 std::string test_time;534 std::string build_commit;535 std::string backend_name;536 std::string op_name;537 std::string op_params;538 std::string test_mode;539 bool supported;540 bool passed;541 std::string error_message;542 double time_us;543 double flops;544 double bandwidth_gb_s;545 size_t memory_kb;546 int n_runs;547 std::string device_description;548 std::string backend_reg_name;549 550 test_result() {551 // Initialize with default values552 time_us = 0.0;553 flops = 0.0;554 bandwidth_gb_s = 0.0;555 memory_kb = 0;556 n_runs = 0;557 supported = false;558 passed = false;559 560 test_time = test_time_now();561 562 // Set build info563 build_commit = ggml_commit();564 }565 566 test_result(const std::string & backend_name, const std::string & op_name, const std::string & op_params,567 const std::string & test_mode, bool supported, bool passed, const std::string & error_message = "",568 double time_us = 0.0, double flops = 0.0, double bandwidth_gb_s = 0.0, size_t memory_kb = 0,569 int n_runs = 0, const std::string & device_description = "", const std::string & backend_reg_name = "") :570 backend_name(backend_name),571 op_name(op_name),572 op_params(op_params),573 test_mode(test_mode),574 supported(supported),575 passed(passed),576 error_message(error_message),577 time_us(time_us),578 flops(flops),579 bandwidth_gb_s(bandwidth_gb_s),580 memory_kb(memory_kb),581 n_runs(n_runs),582 device_description(device_description),583 backend_reg_name(backend_reg_name) {584 test_time = test_time_now();585 586 // Set build info587 build_commit = ggml_commit();588 }589 590 static const std::vector<std::string> & get_fields() {591 static const std::vector<std::string> fields = {592 "test_time", "build_commit", "backend_name", "op_name", "op_params", "test_mode", "supported",593 "passed", "error_message", "time_us", "flops", "bandwidth_gb_s", "memory_kb", "n_runs",594 "device_description", "backend_reg_name"595 };596 return fields;597 }598 599 enum field_type { STRING, BOOL, INT, FLOAT };600 601 static field_type get_field_type(const std::string & field) {602 if (field == "supported" || field == "passed") {603 return BOOL;604 }605 if (field == "memory_kb" || field == "n_runs") {606 return INT;607 }608 if (field == "time_us" || field == "flops" || field == "bandwidth_gb_s") {609 return FLOAT;610 }611 return STRING;612 }613 614 std::vector<std::string> get_values() const {615 return { test_time,616 build_commit,617 backend_name,618 op_name,619 op_params,620 test_mode,621 std::to_string(supported),622 std::to_string(passed),623 error_message,624 std::to_string(time_us),625 std::to_string(flops),626 std::to_string(bandwidth_gb_s),627 std::to_string(memory_kb),628 std::to_string(n_runs),629 device_description,630 backend_reg_name };631 }632};633 634// Printer classes for different output formats635enum class test_status_t { NOT_SUPPORTED, OK, FAIL, SKIPPED };636 637struct test_operation_info {638 std::string op_name;639 std::string op_params;640 std::string backend_name;641 test_status_t status = test_status_t::OK;642 std::string failure_reason;643 644 // Additional information fields that were previously in separate structs645 std::string error_component;646 std::string error_details;647 648 // Gradient info649 int64_t gradient_index = -1;650 std::string gradient_param_name;651 float gradient_value = 0.0f;652 653 // MAA error info654 double maa_error = 0.0;655 double maa_threshold = 0.0;656 657 // Flags for different types of information658 bool has_error = false;659 bool has_gradient_info = false;660 bool has_maa_error = false;661 bool is_compare_failure = false;662 bool is_large_tensor_skip = false;663 664 test_operation_info() = default;665 666 test_operation_info(const std::string & op_name, const std::string & op_params, const std::string & backend_name,667 test_status_t status = test_status_t::OK, const std::string & failure_reason = "") :668 op_name(op_name),669 op_params(op_params),670 backend_name(backend_name),671 status(status),672 failure_reason(failure_reason) {}673 674 // Set error information675 void set_error(const std::string & component, const std::string & details) {676 has_error = true;677 error_component = component;678 error_details = details;679 if (status == test_status_t::OK) {680 status = test_status_t::FAIL;681 }682 }683 684 // Set gradient information685 void set_gradient_info(int64_t index, const std::string & param_name, float value) {686 has_gradient_info = true;687 gradient_index = index;688 gradient_param_name = param_name;689 gradient_value = value;690 if (status == test_status_t::OK) {691 status = test_status_t::FAIL;692 }693 }694 695 // Set MAA error information696 void set_maa_error(double error, double threshold) {697 has_maa_error = true;698 maa_error = error;699 maa_threshold = threshold;700 if (status == test_status_t::OK) {701 status = test_status_t::FAIL;702 }703 }704 705 // Set compare failure706 void set_compare_failure() {707 is_compare_failure = true;708 if (status == test_status_t::OK) {709 status = test_status_t::FAIL;710 }711 }712 713 // Set large tensor skip714 void set_large_tensor_skip() { is_large_tensor_skip = true; }715};716 717struct test_summary_info {718 size_t tests_passed;719 size_t tests_total;720 bool is_backend_summary = false; // true for backend summary, false for test summary721 722 test_summary_info() = default;723 724 test_summary_info(size_t tests_passed, size_t tests_total, bool is_backend_summary = false) :725 tests_passed(tests_passed),726 tests_total(tests_total),727 is_backend_summary(is_backend_summary) {}728};729 730struct testing_start_info {731 size_t device_count;732 733 testing_start_info() = default;734 735 testing_start_info(size_t device_count) : device_count(device_count) {}736};737 738struct backend_init_info {739 size_t device_index;740 size_t total_devices;741 std::string device_name;742 bool skipped = false;743 std::string skip_reason;744 std::string description;745 size_t memory_total_mb = 0;746 size_t memory_free_mb = 0;747 bool has_memory_info = false;748 749 backend_init_info() = default;750 751 backend_init_info(size_t device_index, size_t total_devices, const std::string & device_name, bool skipped = false,752 const std::string & skip_reason = "", const std::string & description = "",753 size_t memory_total_mb = 0, size_t memory_free_mb = 0, bool has_memory_info = false) :754 device_index(device_index),755 total_devices(total_devices),756 device_name(device_name),757 skipped(skipped),758 skip_reason(skip_reason),759 description(description),760 memory_total_mb(memory_total_mb),761 memory_free_mb(memory_free_mb),762 has_memory_info(has_memory_info) {}763};764 765struct backend_status_info {766 std::string backend_name;767 test_status_t status;768 769 backend_status_info() = default;770 771 backend_status_info(const std::string & backend_name, test_status_t status) :772 backend_name(backend_name),773 status(status) {}774};775 776struct overall_summary_info {777 size_t backends_passed;778 size_t backends_total;779 bool all_passed;780 781 overall_summary_info() = default;782 783 overall_summary_info(size_t backends_passed, size_t backends_total, bool all_passed) :784 backends_passed(backends_passed),785 backends_total(backends_total),786 all_passed(all_passed) {}787};788 789struct printer {790 virtual ~printer() {}791 792 FILE * fout = stdout;793 794 virtual void print_header() {}795 796 virtual void print_test_result(const test_result & result) = 0;797 798 virtual void print_footer() {}799 800 virtual void print_operation(const test_operation_info & info) { (void) info; }801 802 virtual void print_summary(const test_summary_info & info) { (void) info; }803 804 virtual void print_testing_start(const testing_start_info & info) { (void) info; }805 806 virtual void print_backend_init(const backend_init_info & info) { (void) info; }807 808 virtual void print_backend_status(const backend_status_info & info) { (void) info; }809 810 virtual void print_overall_summary(const overall_summary_info & info) { (void) info; }811 812 virtual void print_failed_tests(const std::vector<std::string> & failed_tests) { (void) failed_tests; }813};814 815struct console_printer : public printer {816 void print_test_result(const test_result & result) override {817 if (result.test_mode == "test") {818 print_test_console(result);819 } else if (result.test_mode == "perf") {820 print_perf_console(result);821 } else if (result.test_mode == "support") {822 print_support_console(result);823 }824 }825 826 void print_operation(const test_operation_info & info) override {827 printf(" %s(%s): ", info.op_name.c_str(), info.op_params.c_str());828 fflush(stdout);829 830 // Handle large tensor skip first831 if (info.is_large_tensor_skip) {832 printf("skipping large tensors for speed \n");833 return;834 }835 836 // Handle not supported status837 if (info.status == test_status_t::NOT_SUPPORTED) {838 if (!info.failure_reason.empty()) {839 printf("not supported [%s]\n", info.failure_reason.c_str());840 } else {841 printf("not supported [%s]\n", info.backend_name.c_str());842 }843 return;844 }845 846 // Handle errors and additional information847 if (info.has_error) {848 if (info.error_component == "allocation") {849 fprintf(stderr, "failed to allocate tensors [%s] ", info.backend_name.c_str());850 } else if (info.error_component == "backend") {851 fprintf(stderr, " Failed to initialize %s backend\n", info.backend_name.c_str());852 } else {853 fprintf(stderr, "Error in %s: %s\n", info.error_component.c_str(), info.error_details.c_str());854 }855 }856 857 // Handle gradient info858 if (info.has_gradient_info) {859 printf("[%s] nonfinite gradient at index %" PRId64 " (%s=%f) ", info.op_name.c_str(), info.gradient_index,860 info.gradient_param_name.c_str(), info.gradient_value);861 }862 863 // Handle MAA error864 if (info.has_maa_error) {865 printf("[%s] MAA = %.9f > %.9f ", info.op_name.c_str(), info.maa_error, info.maa_threshold);866 }867 868 // Handle compare failure869 if (info.is_compare_failure) {870 printf("compare failed ");871 }872 873 // Print final status874 if (info.status == test_status_t::OK) {875 printf("\033[1;32mOK\033[0m\n");876 } else {877 printf("\033[1;31mFAIL\033[0m\n");878 }879 }880 881 void print_summary(const test_summary_info & info) override {882 if (info.is_backend_summary) {883 printf("%zu/%zu backends passed\n", info.tests_passed, info.tests_total);884 } else {885 printf(" %zu/%zu tests passed\n", info.tests_passed, info.tests_total);886 }887 }888 889 void print_backend_status(const backend_status_info & info) override {890 printf(" Backend %s: ", info.backend_name.c_str());891 if (info.status == test_status_t::OK) {892 printf("\033[1;32mOK\033[0m\n");893 } else {894 printf("\033[1;31mFAIL\033[0m\n");895 }896 }897 898 void print_testing_start(const testing_start_info & info) override {899 printf("Testing %zu devices\n\n", info.device_count);900 }901 902 void print_backend_init(const backend_init_info & info) override {903 printf("Backend %zu/%zu: %s\n", info.device_index + 1, info.total_devices, info.device_name.c_str());904 905 if (info.skipped) {906 printf(" %s\n", info.skip_reason.c_str());907 return;908 }909 910 if (!info.description.empty()) {911 printf(" Device description: %s\n", info.description.c_str());912 }913 914 if (info.has_memory_info) {915 printf(" Device memory: %zu MB (%zu MB free)\n", info.memory_total_mb, info.memory_free_mb);916 }917 918 printf("\n");919 }920 921 void print_overall_summary(const overall_summary_info & info) override {922 printf("%zu/%zu backends passed\n", info.backends_passed, info.backends_total);923 if (info.all_passed) {924 printf("\033[1;32mOK\033[0m\n");925 } else {926 printf("\033[1;31mFAIL\033[0m\n");927 }928 }929 930 void print_failed_tests(const std::vector<std::string> & failed_tests) override {931 if (failed_tests.empty()) {932 return;933 }934 935 printf("\nFailing tests:\n");936 for (const auto & test_name : failed_tests) {937 printf(" %s\n", test_name.c_str());938 }939 }940 941 private:942 void print_test_console(const test_result & result) {943 printf(" %s(%s): ", result.op_name.c_str(), result.op_params.c_str());944 fflush(stdout);945 946 if (!result.supported) {947 printf("not supported [%s] ", result.backend_name.c_str());948 printf("\n");949 return;950 }951 952 if (result.passed) {953 printf("\033[1;32mOK\033[0m\n");954 } else {955 printf("\033[1;31mFAIL\033[0m\n");956 }957 }958 959 void print_perf_console(const test_result & result) {960 int len = printf(" %s(%s): ", result.op_name.c_str(), result.op_params.c_str());961 fflush(stdout);962 963 if (!result.supported) {964 printf("not supported\n");965 return;966 }967 968 // align while also leaving some margin for variations in parameters969 int align = 8;970 int last = (len + align - 1) / align * align;971 if (last - len < 5) {972 last += align;973 }974 printf("%*s", last - len, "");975 976 printf(" %8d runs - %8.2f us/run - ", result.n_runs, result.time_us);977 978 if (result.flops > 0) {979 auto format_flops = [](double flops) -> std::string {980 char buf[256];981 if (flops >= 1e12) {982 snprintf(buf, sizeof(buf), "%6.2f TFLOP", flops / 1e12);983 } else if (flops >= 1e9) {984 snprintf(buf, sizeof(buf), "%6.2f GFLOP", flops / 1e9);985 } else if (flops >= 1e6) {986 snprintf(buf, sizeof(buf), "%6.2f MFLOP", flops / 1e6);987 } else {988 snprintf(buf, sizeof(buf), "%6.2f kFLOP", flops / 1e3);989 }990 return buf;991 };992 uint64_t op_flops_per_run = result.flops * result.time_us / 1e6;993 printf("%s/run - \033[1;34m%sS\033[0m", format_flops(op_flops_per_run).c_str(),994 format_flops(result.flops).c_str());995 } else {996 printf("%8zu kB/run - \033[1;34m%7.2f GB/s\033[0m", result.memory_kb, result.bandwidth_gb_s);997 }998 printf("\n");999 }1000 1001 void print_support_console(const test_result & result) {1002 printf(" %s(%s): ", result.op_name.c_str(), result.op_params.c_str());1003 fflush(stdout);1004 1005 if (result.supported) {1006 printf("\033[1;32mSUPPORTED\033[0m\n");1007 } else {1008 printf("\033[1;31mNOT SUPPORTED\033[0m\n");1009 }1010 }1011};1012 1013struct sql_printer : public printer {1014 static std::string get_sql_field_type(const std::string & field) {1015 switch (test_result::get_field_type(field)) {1016 case test_result::STRING:1017 return "TEXT";1018 case test_result::BOOL:1019 case test_result::INT:1020 return "INTEGER";1021 case test_result::FLOAT:1022 return "REAL";1023 default:1024 GGML_ABORT("invalid field type");1025 }1026 }1027 1028 void print_header() override {1029 std::vector<std::string> fields = test_result::get_fields();1030 fprintf(fout, "CREATE TABLE IF NOT EXISTS test_backend_ops (\n");1031 for (size_t i = 0; i < fields.size(); i++) {1032 fprintf(fout, " %s %s%s\n", fields[i].c_str(), get_sql_field_type(fields[i]).c_str(),1033 i < fields.size() - 1 ? "," : "");1034 }1035 fprintf(fout, ");\n\n");1036 }1037 1038 void print_test_result(const test_result & result) override {1039 fprintf(fout, "INSERT INTO test_backend_ops (");1040 std::vector<std::string> fields = test_result::get_fields();1041 for (size_t i = 0; i < fields.size(); i++) {1042 fprintf(fout, "%s%s", fields[i].c_str(), i < fields.size() - 1 ? ", " : "");1043 }1044 fprintf(fout, ") VALUES (");1045 std::vector<std::string> values = result.get_values();1046 for (size_t i = 0; i < values.size(); i++) {1047 fprintf(fout, "'%s'%s", values[i].c_str(), i < values.size() - 1 ? ", " : "");1048 }1049 fprintf(fout, ");\n");1050 }1051};1052 1053struct csv_printer : public printer {1054 void print_header() override {1055 1056 std::vector<std::string> fields = test_result::get_fields();1057 std::vector<std::string> fields_csv = get_fields_csv();1058 for (size_t i = 0; i < fields.size(); i++) {1059 if (std::find(std::begin(fields_csv), std::end(fields_csv), fields[i]) == std::end(fields_csv)) {1060 continue;1061 }1062 printf("\"%s\"%s", fields[i].c_str(), i < fields.size() - 1 ? "," : "");1063 }1064 printf("\n");1065 }1066 1067 void print_test_result(const test_result & result) override {1068 1069 std::vector<std::string> values = result.get_values();1070 std::vector<std::string> fields = test_result::get_fields();1071 std::vector<std::string> fields_csv = get_fields_csv();1072 1073 for (size_t i = 0; i < values.size(); i++) {1074 1075 if (std::find(std::begin(fields_csv), std::end(fields_csv), fields[i]) == std::end(fields_csv)) {1076 continue;1077 }1078 1079 // Escape quotes and wrap in quotes for CSV1080 std::string escaped_value = values[i];1081 size_t pos = 0;1082 while ((pos = escaped_value.find("\"", pos)) != std::string::npos) {1083 escaped_value.replace(pos, 1, "\"\"");1084 pos += 2;1085 }1086 printf("\"%s\"%s", escaped_value.c_str(), i < values.size() - 1 ? "," : "");1087 }1088 printf("\n");1089 }1090 1091 static std::vector<std::string> get_fields_csv() {1092 return {1093 "op_name",1094 "op_params",1095 "supported",1096 "error_message",1097 "test_mode",1098 "backend_reg_name",1099 "backend_name",1100 };1101 }1102 1103};1104 1105static std::unique_ptr<printer> create_printer(output_formats format) {1106 switch (format) {1107 case CONSOLE:1108 return std::make_unique<console_printer>();1109 case SQL:1110 return std::make_unique<sql_printer>();1111 case CSV:1112 return std::make_unique<csv_printer>();1113 }1114 GGML_ABORT("invalid output format");1115}1116 1117static std::mutex g_test_output_mutex;1118 1119static void print_test_result_locked(printer * output_printer, const test_result & result) {1120 if (output_printer == nullptr) {1121 return;1122 }1123 1124 std::lock_guard<std::mutex> guard(g_test_output_mutex);1125 output_printer->print_test_result(result);1126}1127 1128struct test_case {1129 virtual ~test_case() {}1130 1131 virtual std::string op_desc(ggml_tensor * t) {1132 return ggml_op_desc(t);1133 }1134 1135 virtual std::string vars() {1136 return "";1137 }1138 1139 virtual ggml_tensor * build_graph(ggml_context * ctx) = 0;1140 virtual ggml_tensor * build_graph(ggml_context * ctx, ggml_context * ctx_weights) {1141 GGML_UNUSED(ctx_weights);1142 return build_graph(ctx);1143 }1144 1145 virtual double max_nmse_err() {1146 return 1e-7;1147 }1148 1149 virtual double max_nmse_err(ggml_backend_t backend) {1150 ggml_backend_reg_t reg = ggml_backend_dev_backend_reg(ggml_backend_get_device(backend));1151 // See https://github.com/ggml-org/llama.cpp/pull/22976 for explanation.1152 if (contains_f16 && strcmp(ggml_backend_reg_name(reg), "WebGPU") == 0) {1153 return std::max(max_nmse_err(), 1e-6);1154 }1155 return max_nmse_err();1156 }1157 1158 virtual double max_maa_err() {1159 return 1e-4;1160 }1161 1162 virtual double max_err() {1163 return max_nmse_err();1164 }1165 1166 virtual double max_err(ggml_backend_t backend) {1167 return max_nmse_err(backend);1168 }1169 1170 virtual double err(const float * a, const float * b, size_t n) {1171 return nmse(a, b, n);1172 }1173 1174 virtual float grad_eps() {1175 return 1e-1f;1176 }1177 1178 // If false, estimate gradient with 2 points, neglects 3rd order derivative and higher.1179 // If true, estimate gradient with 4 points, neglects 5th order derivative and higher.1180 virtual bool grad_precise() {1181 return false;1182 }1183 1184 // Skip gradient checks if total number of gradients to be checked is larger than this (to speed up the tests).1185 virtual int64_t grad_nmax() {1186 return 10000;1187 }1188 1189 // No effect if empty.1190 // If not empty, skip all gradient checks where the numerical result does not match any of the values.1191 // Needed for dealing with noncontinuous gradients (e.g. ReLU) where estimation using finite differences is unreliable.1192 virtual std::vector<float> grad_expect() {1193 return {};1194 }1195 1196 virtual void initialize_tensors(ggml_context * ctx) {1197 for (ggml_tensor * t = ggml_get_first_tensor(ctx); t != nullptr; t = ggml_get_next_tensor(ctx, t)) {1198 init_tensor_uniform(t);1199 }1200 }