Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#include "ggml-et-cpu-compare.h"2 3#include "ggml-cpu/ggml-cpu-impl.h"4#include "ggml-cpu/ops.h"5 6#include <algorithm>7#include <cmath>8#include <cstdlib>9#include <cstring>10 11bool ggml_et_cpu_compare_init_pre(ggml_et_cpu_compare_ctx * ctx, const ggml_tensor * node, ggml_op op) {12 if (!ctx || !node) {13 GGML_LOG_ERROR("ET: Invalid parameters for CPU compare init\n");14 return false;15 }16 17 // Clear context18 memset(ctx, 0, sizeof(*ctx));19 20 // Calculate actual buffer sizes - use backend buffer size for accurate copy21 auto get_tensor_buffer_size = [](const ggml_tensor * tensor) -> size_t {22 if (!tensor) {23 return 0;24 }25 26 if (tensor->buffer) {27 // Get actual backend buffer size28 size_t buffer_size = ggml_backend_buffer_get_size(tensor->buffer);29 30 // Use the full buffer size to avoid any truncation issues31 return buffer_size;32 } else {33 // Fallback to logical size if no buffer34 return ggml_nbytes(tensor);35 }36 };37 38 ctx->src0_size = get_tensor_buffer_size(node->src[0]);39 ctx->src1_size = get_tensor_buffer_size(node->src[1]);40 ctx->src2_size = get_tensor_buffer_size(node->src[2]);41 ctx->dst_size = get_tensor_buffer_size(node);42 43 // Allocate CPU buffers for all tensors44 if (ctx->src0_size > 0) {45 ctx->cpu_src0_data = malloc(ctx->src0_size);46 if (!ctx->cpu_src0_data) {47 GGML_LOG_ERROR("ET: Failed to allocate CPU src0 buffer\n");48 goto cleanup;49 }50 }51 52 if (ctx->src1_size > 0) {53 ctx->cpu_src1_data = malloc(ctx->src1_size);54 if (!ctx->cpu_src1_data) {55 GGML_LOG_ERROR("ET: Failed to allocate CPU src1 buffer\n");56 goto cleanup;57 }58 }59 60 if (ctx->src2_size > 0) {61 ctx->cpu_src2_data = malloc(ctx->src2_size);62 if (!ctx->cpu_src2_data) {63 GGML_LOG_ERROR("ET: Failed to allocate CPU src2 buffer\n");64 goto cleanup;65 }66 }67 68 ctx->cpu_dst_data = malloc(ctx->dst_size);69 if (!ctx->cpu_dst_data) {70 GGML_LOG_ERROR("ET: Failed to allocate CPU dst buffer\n");71 goto cleanup;72 }73 74 ctx->et_dst_data = malloc(ctx->dst_size);75 if (!ctx->et_dst_data) {76 GGML_LOG_ERROR("ET: Failed to allocate ET dst buffer\n");77 goto cleanup;78 }79 80 // Copy data from ET device buffers to CPU host buffers81 if (ctx->src0_size > 0) {82 // Copy logical tensor size - ggml_backend_tensor_get handles stride layout internally83 size_t logical_size = ggml_nbytes(node->src[0]);84 ggml_backend_tensor_get(node->src[0], ctx->cpu_src0_data, 0, logical_size);85 }86 if (ctx->src1_size > 0) {87 size_t logical_size = ggml_nbytes(node->src[1]);88 ggml_backend_tensor_get(node->src[1], ctx->cpu_src1_data, 0, logical_size);89 }90 if (ctx->src2_size > 0) {91 size_t logical_size = ggml_nbytes(node->src[2]);92 ggml_backend_tensor_get(node->src[2], ctx->cpu_src2_data, 0, logical_size);93 }94 95 // Copy destination data from device (for operations like SET_ROWS that modify existing data)96 // Most ops create new tensors so this is unused, but SET_ROWS requires existing dst data97 {98 size_t logical_size = ggml_nbytes(node);99 ggml_backend_tensor_get(node, ctx->cpu_dst_data, 0, logical_size);100 }101 102 // Create CPU backend for reference computation103 GGML_LOG_DEBUG("ET: Creating CPU backend for reference computation\n");104 ctx->cpu_backend = ggml_backend_cpu_init();105 if (!ctx->cpu_backend) {106 GGML_LOG_ERROR("ET: Failed to create CPU backend\n");107 goto cleanup;108 }109 110 // Create GGML context for CPU tensors111 GGML_LOG_DEBUG("ET: Creating GGML context for CPU computation\n");112 ggml_init_params ctx_params;113 ctx_params.mem_size = ggml_tensor_overhead() * 4 + ggml_graph_overhead(); // up to 4 tensors + graph114 ctx_params.mem_buffer = nullptr;115 ctx_params.no_alloc = true; // We'll manage data ourselves116 ctx->ggml_ctx = ggml_init(ctx_params);117 if (!ctx->ggml_ctx) {118 GGML_LOG_ERROR("ET: Failed to create GGML context\n");119 goto cleanup;120 }121 122 // Create CPU tensors with proper context123 if (node->src[0]) {124 ctx->cpu_src0 = ggml_new_tensor(ctx->ggml_ctx, node->src[0]->type, GGML_MAX_DIMS, node->src[0]->ne);125 if (!ctx->cpu_src0) {126 GGML_LOG_ERROR("ET: Failed to create CPU src0 tensor\n");127 goto cleanup;128 }129 ctx->cpu_src0->data = ctx->cpu_src0_data;130 // Copy stride array (nb) for correct memory layout131 memcpy(ctx->cpu_src0->nb, node->src[0]->nb, sizeof(node->src[0]->nb));132 // Copy op_params if present133 memcpy(ctx->cpu_src0->op_params, node->src[0]->op_params, sizeof(node->src[0]->op_params));134 }135 136 if (node->src[1]) {137 ctx->cpu_src1 = ggml_new_tensor(ctx->ggml_ctx, node->src[1]->type, GGML_MAX_DIMS, node->src[1]->ne);138 if (!ctx->cpu_src1) {139 GGML_LOG_ERROR("ET: Failed to create CPU src1 tensor\n");140 goto cleanup;141 }142 ctx->cpu_src1->data = ctx->cpu_src1_data;143 // Copy stride array (nb) for correct memory layout144 memcpy(ctx->cpu_src1->nb, node->src[1]->nb, sizeof(node->src[1]->nb));145 // Copy op_params if present146 memcpy(ctx->cpu_src1->op_params, node->src[1]->op_params, sizeof(node->src[1]->op_params));147 }148 149 if (node->src[2]) {150 ctx->cpu_src2 = ggml_new_tensor(ctx->ggml_ctx, node->src[2]->type, GGML_MAX_DIMS, node->src[2]->ne);151 if (!ctx->cpu_src2) {152 GGML_LOG_ERROR("ET: Failed to create CPU src2 tensor\n");153 goto cleanup;154 }155 ctx->cpu_src2->data = ctx->cpu_src2_data;156 // Copy stride array (nb) for correct memory layout157 memcpy(ctx->cpu_src2->nb, node->src[2]->nb, sizeof(node->src[2]->nb));158 // Copy op_params if present159 memcpy(ctx->cpu_src2->op_params, node->src[2]->op_params, sizeof(node->src[2]->op_params));160 }161 162 return true;163 164cleanup:165 ggml_et_cpu_compare_free(ctx);166 return false;167}168 169bool ggml_et_cpu_compare_compute_and_check(ggml_et_cpu_compare_ctx * ctx,170 const ggml_tensor * node,171 const ggml_et_cpu_compare_config * config) {172 if (!ctx || !ctx->cpu_backend || !ctx->ggml_ctx || !node || !config) {173 GGML_LOG_ERROR("ET: Invalid parameters for CPU compute and check\n");174 return false;175 }176 177 // Create operation-specific CPU destination tensor based on the node's operation178 ggml_op op = node->op;179 switch (op) {180 case GGML_OP_MUL:181 ctx->cpu_dst = ggml_mul(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1);182 break;183 case GGML_OP_ADD:184 ctx->cpu_dst = ggml_add(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1);185 break;186 case GGML_OP_MUL_MAT:187 ctx->cpu_dst = ggml_mul_mat(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1);188 break;189 case GGML_OP_MUL_MAT_ID:190 // MUL_MAT_ID: Mixture of Experts matrix multiplication191 // src0 (as): expert weight matrices [K, M, n_expert]192 // src1 (b): activations [K, n_expert_used, batch]193 // src2 (ids): expert selection indices [n_expert_used, batch]194 ctx->cpu_dst = ggml_mul_mat_id(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, ctx->cpu_src2);195 break;196 case GGML_OP_ROPE:197 {198 const int32_t * op_params = (const int32_t *) node->op_params;199 const int32_t n_dims = op_params[1];200 const int32_t mode = op_params[2];201 const int32_t n_ctx_orig = op_params[4];202 const float freq_base = *((const float *) (op_params + 5));203 const float freq_scale = *((const float *) (op_params + 6));204 const float ext_factor = *((const float *) (op_params + 7));205 const float attn_factor = *((const float *) (op_params + 8));206 const float beta_fast = *((const float *) (op_params + 9));207 const float beta_slow = *((const float *) (op_params + 10));208 209 if (mode & GGML_ROPE_TYPE_MROPE) {210 int sections[GGML_MROPE_SECTIONS];211 memcpy(sections, op_params + 11, sizeof(sections));212 ctx->cpu_dst = ggml_rope_multi(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, ctx->cpu_src2,213 n_dims, sections, mode, n_ctx_orig, freq_base, freq_scale,214 ext_factor, attn_factor, beta_fast, beta_slow);215 } else {216 ctx->cpu_dst = ggml_rope_ext(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, ctx->cpu_src2,217 n_dims, mode, n_ctx_orig, freq_base, freq_scale, ext_factor,218 attn_factor, beta_fast, beta_slow);219 }220 }221 break;222 case GGML_OP_RMS_NORM:223 // Extract epsilon parameter from op_params (stored as float)224 {225 float eps;226 memcpy(&eps, node->op_params, sizeof(float));227 ctx->cpu_dst = ggml_rms_norm(ctx->ggml_ctx, ctx->cpu_src0, eps);228 }229 break;230 case GGML_OP_SQR:231 ctx->cpu_dst = ggml_sqr(ctx->ggml_ctx, ctx->cpu_src0);232 break;233 case GGML_OP_UNARY:234 {235 ggml_unary_op uop = (ggml_unary_op) ggml_get_op_params_i32(node, 0);236 ctx->cpu_dst = ggml_unary(ctx->ggml_ctx, ctx->cpu_src0, uop);237 }238 break;239 case GGML_OP_SUM_ROWS:240 ctx->cpu_dst = ggml_sum_rows(ctx->ggml_ctx, ctx->cpu_src0);241 break;242 case GGML_OP_MEAN:243 ctx->cpu_dst = ggml_mean(ctx->ggml_ctx, ctx->cpu_src0);244 break;245 case GGML_OP_CLAMP:246 {247 float clamp_min, clamp_max;248 memcpy(&clamp_min, (const float *) node->op_params + 0, sizeof(float));249 memcpy(&clamp_max, (const float *) node->op_params + 1, sizeof(float));250 ctx->cpu_dst = ggml_clamp(ctx->ggml_ctx, ctx->cpu_src0, clamp_min, clamp_max);251 }252 break;253 case GGML_OP_GLU:254 // Extract GLU parameters from op_params (split mode only)255 {256 int32_t glu_op_type = ggml_get_op_params_i32(node, 0); // GLU variant257 ggml_glu_op glu_op = (ggml_glu_op) glu_op_type;258 259 // Only support split tensor mode260 if (!ctx->cpu_src1) {261 GGML_LOG_ERROR("ET: GLU CPU comparison requires split tensor mode\n");262 return false;263 }264 ctx->cpu_dst = ggml_glu_split(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, glu_op);265 }266 break;267 case GGML_OP_SOFT_MAX:268 {269 // Extract scale and max_bias from op_params270 float scale = 1.0f;271 float max_bias = 0.0f;272 memcpy(&scale, (const float *) node->op_params + 0, sizeof(float));273 memcpy(&max_bias, (const float *) node->op_params + 1, sizeof(float));274 275 if (ctx->cpu_src1 || scale != 1.0f || max_bias != 0.0f) {276 // Use extended softmax when mask or non-default parameters are present277 ctx->cpu_dst = ggml_soft_max_ext(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, scale, max_bias);278 } else {279 // Use simple softmax when no mask and default parameters280 ctx->cpu_dst = ggml_soft_max(ctx->ggml_ctx, ctx->cpu_src0);281 }282 283 // Add sinks if present284 if (ctx->cpu_src2) {285 ggml_soft_max_add_sinks(ctx->cpu_dst, ctx->cpu_src2);286 }287 }288 break;289 case GGML_OP_GET_ROWS:290 ctx->cpu_dst = ggml_get_rows(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1);291 break;292 case GGML_OP_CONT:293 ctx->cpu_dst = ggml_cont(ctx->ggml_ctx, ctx->cpu_src0);294 break;295 case GGML_OP_SET_ROWS:296 {297 // SET_ROWS operation scatters src0 rows to dst[src1] positions298 // Create destination tensor (this is the "view" that SET_ROWS returns)299 ggml_tensor * cpu_dst_base = ggml_new_tensor(ctx->ggml_ctx, node->type, GGML_MAX_DIMS, node->ne);300 if (!cpu_dst_base) {301 GGML_LOG_ERROR("ET: Failed to create CPU destination base tensor for SET_ROWS\n");302 return false;303 }304 cpu_dst_base->data = ctx->cpu_dst_data;305 memcpy(cpu_dst_base->nb, node->nb, sizeof(node->nb));306 307 // Note: cpu_dst_data already contains the pre-existing destination data from device308 // SET_ROWS will update specific rows, leaving others unchanged309 310 // Perform SET_ROWS operation: returns a view that scatters src0 rows to dst[src1] positions311 ctx->cpu_dst = ggml_set_rows(ctx->ggml_ctx, cpu_dst_base, ctx->cpu_src0, ctx->cpu_src1);312 }313 break;314 default:315 GGML_LOG_ERROR("ET: Unsupported operation %s for CPU comparison\n", ggml_op_name(op));316 return false;317 }318 319 if (!ctx->cpu_dst) {320 GGML_LOG_ERROR("ET: Failed to create CPU destination tensor for operation %s\n", ggml_op_name(op));321 return false;322 }323 324 ctx->cpu_dst->data = ctx->cpu_dst_data;325 // Copy stride array (nb) for correct memory layout - except for CONT which should keep contiguous strides326 if (op != GGML_OP_CONT) {327 memcpy(ctx->cpu_dst->nb, node->nb, sizeof(node->nb));328 }329 // For CONT operations, keep the contiguous strides created by ggml_cont()330 331 // Create minimal computation graph332 ctx->cpu_graph = ggml_new_graph_custom(ctx->ggml_ctx, 1, false);333 if (!ctx->cpu_graph) {334 GGML_LOG_ERROR("ET: Failed to create CPU computation graph\n");335 return false;336 }337 ctx->cpu_graph->nodes[0] = ctx->cpu_dst;338 ctx->cpu_graph->n_nodes = 1;339 340 // Log input data for debugging if enabled341 if (config && config->log_differences) {342 if (ctx->cpu_src0_data && ctx->src0_size >= 4) {343 GGML_LOG_DEBUG("ET: CPU src0 first few bytes: %02x %02x %02x %02x\n", ((uint8_t *) ctx->cpu_src0_data)[0],344 ((uint8_t *) ctx->cpu_src0_data)[1], ((uint8_t *) ctx->cpu_src0_data)[2],345 ((uint8_t *) ctx->cpu_src0_data)[3]);346 }347 if (ctx->cpu_src1_data && ctx->src1_size >= 16) {348 GGML_LOG_DEBUG("ET: CPU src1 first few floats: %.6f %.6f %.6f %.6f\n", ((float *) ctx->cpu_src1_data)[0],349 ((float *) ctx->cpu_src1_data)[1], ((float *) ctx->cpu_src1_data)[2],350 ((float *) ctx->cpu_src1_data)[3]);351 }352 }353 354 // Compute using CPU backend355 ggml_status cpu_result = ggml_backend_graph_compute(ctx->cpu_backend, ctx->cpu_graph);356 357 if (cpu_result != GGML_STATUS_SUCCESS) {358 GGML_LOG_ERROR("ET: CPU reference computation failed with status %d\n", cpu_result);359 return false;360 }361 362 // Log output data for debugging if enabled363 if (config && config->log_differences && ctx->dst_size >= 16) {364 GGML_LOG_DEBUG("ET: CPU dst first few floats after computation: %.6f %.6f %.6f %.6f\n",365 ((float *) ctx->cpu_dst_data)[0], ((float *) ctx->cpu_dst_data)[1],366 ((float *) ctx->cpu_dst_data)[2], ((float *) ctx->cpu_dst_data)[3]);367 }368 369 // Now copy ET device destination to host for comparison370 size_t dst_logical_size = ggml_nbytes(node);371 ggml_backend_tensor_get(node, ctx->et_dst_data, 0, dst_logical_size);372 373 if (config->log_differences) {374 size_t num_elements = ggml_nelements(node);375 size_t max_log = std::min(num_elements, config->max_log_elements);376 377 // Check if this is an elementwise operation that can show src inputs378 bool is_elementwise = (op == GGML_OP_MUL || op == GGML_OP_ADD || op == GGML_OP_GLU);379 float * cpu_src0_float = is_elementwise ? (float *) ctx->cpu_src0_data : nullptr;380 float * cpu_src1_float = is_elementwise ? (float *) ctx->cpu_src1_data : nullptr;381 382 // Helper to get float value from tensor data (handles f16 and f32)383 auto get_float = [](const void * data, size_t idx, ggml_type type) -> float {384 if (type == GGML_TYPE_F16) {385 const ggml_fp16_t * fp16_data = (const ggml_fp16_t *) data;386 return ggml_fp16_to_fp32(fp16_data[idx]);387 }388 389 const float * float_data = (const float *) data;390 return float_data[idx];391 };392 393 // Compare all elements but log only the first max_log_elements394 bool matches = true;395 size_t total_mismatches = 0;396 397 // First pass: check all elements for mismatches398 for (size_t i = 0; i < num_elements; i++) {399 float cpu_val = get_float(ctx->cpu_dst_data, i, node->type);400 float et_val = get_float(ctx->et_dst_data, i, node->type);401 float diff = fabsf(cpu_val - et_val);402 float rel_diff = diff / (fabsf(cpu_val) + 1e-8f);403 404 if (rel_diff > config->tolerance) {405 matches = false;406 total_mismatches++;407 }408 }409 410 // Second pass: log detailed info for first max_log elements only411 for (size_t i = 0; i < max_log; i++) {412 float cpu_val = get_float(ctx->cpu_dst_data, i, node->type);413 float et_val = get_float(ctx->et_dst_data, i, node->type);414 float diff = fabsf(cpu_val - et_val);415 416 if (is_elementwise && cpu_src0_float && cpu_src1_float) {417 GGML_LOG_DEBUG("ET: [%zu] src0=%.6f, src1=%.6f -> CPU=%.6f, ET=%.6f, diff=%.6f\n", i, cpu_src0_float[i],418 cpu_src1_float[i], cpu_val, et_val, diff);419 } else if (is_elementwise && cpu_src0_float) {420 GGML_LOG_DEBUG("ET: [%zu] src0=%.6f -> CPU=%.6f, ET=%.6f, diff=%.6f\n", i, cpu_src0_float[i], cpu_val,421 et_val, diff);422 } else {423 GGML_LOG_DEBUG("ET: [%zu] CPU=%.6f, ET=%.6f, diff=%.6f\n", i, cpu_val, et_val, diff);424 }425 }426 427 // Check some elements from the middle and end for full coverage428 if (num_elements > max_log) {429 size_t mid = num_elements / 2;430 size_t end = num_elements - 1;431 float cpu_mid = get_float(ctx->cpu_dst_data, mid, node->type);432 float et_mid = get_float(ctx->et_dst_data, mid, node->type);433 float cpu_end = get_float(ctx->cpu_dst_data, end, node->type);434 float et_end = get_float(ctx->et_dst_data, end, node->type);435 436 GGML_LOG_DEBUG("ET: Middle element [%zu]: CPU=%.6f, ET=%.6f\n", mid, cpu_mid, et_mid);437 GGML_LOG_DEBUG("ET: Last element [%zu]: CPU=%.6f, ET=%.6f\n", end, cpu_end, et_end);438 }439 440 GGML_LOG_DEBUG("ET: Results %s (%zu/%zu elements match within tolerance %.6f)\n", matches ? "MATCH" : "DIFFER",441 num_elements - total_mismatches, num_elements, config->tolerance);442 }443 444 // Copy CPU result to device if flag is set445 if (config->use_cpu_result) {446 GGML_LOG_DEBUG("ET: Overwriting ET device result with CPU result for correct inference\n");447 size_t dst_logical_size = ggml_nbytes(node);448 ggml_backend_tensor_set(const_cast<ggml_tensor *>(node), ctx->cpu_dst_data, 0, dst_logical_size);449 GGML_LOG_DEBUG("ET: CPU result copied to ET device buffer\n");450 }451 452 return true;453}454 455void ggml_et_cpu_compare_free(ggml_et_cpu_compare_ctx * ctx) {456 if (!ctx) {457 return;458 }459 460 if (ctx->cpu_src0_data) {461 free(ctx->cpu_src0_data);462 ctx->cpu_src0_data = nullptr;463 }464 if (ctx->cpu_src1_data) {465 free(ctx->cpu_src1_data);466 ctx->cpu_src1_data = nullptr;467 }468 if (ctx->cpu_src2_data) {469 free(ctx->cpu_src2_data);470 ctx->cpu_src2_data = nullptr;471 }472 if (ctx->cpu_dst_data) {473 free(ctx->cpu_dst_data);474 ctx->cpu_dst_data = nullptr;475 }476 if (ctx->et_dst_data) {477 free(ctx->et_dst_data);478 ctx->et_dst_data = nullptr;479 }480 481 if (ctx->ggml_ctx) {482 ggml_free(ctx->ggml_ctx);483 ctx->ggml_ctx = nullptr;484 }485 486 if (ctx->cpu_backend) {487 ggml_backend_free(ctx->cpu_backend);488 ctx->cpu_backend = nullptr;489 }490 491 // Clear pointers492 ctx->cpu_src0 = nullptr;493 ctx->cpu_src1 = nullptr;494 ctx->cpu_src2 = nullptr;495 ctx->cpu_dst = nullptr;496 ctx->cpu_graph = nullptr;497}498 