Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#include "ggml-et-ops.h"2 3#include "ggml-et-cpu-compare.h"4#include "ggml-et-kernels.h"5#include "ggml-impl.h"6 7#include <stdio.h>8 9#include <cstdint>10 11// CPU comparison configuration - can be enabled for debugging12static ggml_et_cpu_compare_config rope_cpu_compare_config = {13 /* .enabled = */ false,14 /* .use_cpu_result = */ false, // Replace ET result with CPU result15 /* .log_differences = */ true,16 /* .tolerance = */ 1e-5f,17 /* .max_log_elements = */ 409618};19 20static ggml_et_cpu_compare_config rms_norm_cpu_compare_config = {21 /* .enabled = */ false,22 /* .use_cpu_result = */ false,23 /* .log_differences = */ true,24 /* .tolerance = */ 1e-5f,25 /* .max_log_elements = */ 409626};27 28static ggml_et_cpu_compare_config norm_cpu_compare_config = {29 /* .enabled = */ false,30 /* .use_cpu_result = */ false,31 /* .log_differences = */ true,32 /* .tolerance = */ 1e-5f,33 /* .max_log_elements = */ 409634};35 36static ggml_et_cpu_compare_config l2_norm_cpu_compare_config = {37 /* .enabled = */ false,38 /* .use_cpu_result = */ false,39 /* .log_differences = */ true,40 /* .tolerance = */ 1e-5f,41 /* .max_log_elements = */ 409642};43 44static ggml_et_cpu_compare_config group_norm_cpu_compare_config = {45 /* .enabled = */ false,46 /* .use_cpu_result = */ false,47 /* .log_differences = */ true,48 /* .tolerance = */ 1e-5f,49 /* .max_log_elements = */ 409650};51 52static ggml_et_cpu_compare_config im2col_cpu_compare_config = {53 /* .enabled = */ false,54 /* .use_cpu_result = */ false,55 /* .log_differences = */ true,56 /* .tolerance = */ 1e-5f,57 /* .max_log_elements = */ 409658};59 60static ggml_et_cpu_compare_config unary_cpu_compare_config = {61 /* .enabled = */ false,62 /* .use_cpu_result = */ false,63 /* .log_differences = */ true,64 /* .tolerance = */ 1e-4f,65 /* .max_log_elements = */ 409666};67 68static ggml_et_cpu_compare_config sum_rows_cpu_compare_config = {69 /* .enabled = */ false,70 /* .use_cpu_result = */ false,71 /* .log_differences = */ true,72 /* .tolerance = */ 1e-5f,73 /* .max_log_elements = */ 409674};75 76static ggml_et_cpu_compare_config clamp_cpu_compare_config = {77 /* .enabled = */ false,78 /* .use_cpu_result = */ false,79 /* .log_differences = */ true,80 /* .tolerance = */ 1e-6f,81 /* .max_log_elements = */ 409682};83 84static ggml_et_cpu_compare_config mean_cpu_compare_config = {85 /* .enabled = */ false,86 /* .use_cpu_result = */ false,87 /* .log_differences = */ true,88 /* .tolerance = */ 1e-5f,89 /* .max_log_elements = */ 409690};91 92static ggml_et_cpu_compare_config sqr_cpu_compare_config = {93 /* .enabled = */ false,94 /* .use_cpu_result = */ false,95 /* .log_differences = */ true,96 /* .tolerance = */ 1e-6f,97 /* .max_log_elements = */ 409698};99 100static ggml_et_cpu_compare_config elmap_cpu_compare_config = {101 /* .enabled = */ false,102 /* .use_cpu_result = */ false,103 /* .log_differences = */ true,104 /* .tolerance = */ 1e-6f,105 /* .max_log_elements = */ 4096106};107 108static ggml_et_cpu_compare_config glu_cpu_compare_config = {109 /* .enabled = */ false,110 /* .use_cpu_result = */ false,111 /* .log_differences = */ true,112 /* .tolerance = */ 1e-5f,113 /* .max_log_elements = */ 4096114};115 116static ggml_et_cpu_compare_config mul_mat_cpu_compare_config = {117 /* .enabled = */ false,118 /* .use_cpu_result = */ false,119 /* .log_differences = */ true,120 /* .tolerance = */ 0.01,121 /* .max_log_elements = */ 4096122};123 124static ggml_et_cpu_compare_config mul_mat_id_cpu_compare_config = {125 /* .enabled = */ false,126 /* .use_cpu_result = */ false,127 /* .log_differences = */ true,128 /* .tolerance = */ 0.01,129 /* .max_log_elements = */ 4096130};131 132static ggml_et_cpu_compare_config softmax_cpu_compare_config = {133 /* .enabled = */ false,134 /* .use_cpu_result = */ false,135 /* .log_differences = */ true,136 /* .tolerance = */ 1e-5f,137 /* .max_log_elements = */ 1024138};139 140static ggml_et_cpu_compare_config get_rows_cpu_compare_config = {141 /* .enabled = */ false,142 /* .use_cpu_result = */ false,143 /* .log_differences = */ true,144 /* .tolerance = */ 1e-6f,145 /* .max_log_elements = */ 2048146};147 148static ggml_et_cpu_compare_config pad_cpu_compare_config = {149 /* .enabled = */ false,150 /* .use_cpu_result = */ false,151 /* .log_differences = */ true,152 /* .tolerance = */ 1e-6f,153 /* .max_log_elements = */ 4096154};155 156static ggml_et_cpu_compare_config cont_cpu_compare_config = {157 /* .enabled = */ false,158 /* .use_cpu_result = */ false,159 /* .log_differences = */ true,160 /* .tolerance = */ 1e-6f,161 /* .max_log_elements = */ 4096162};163 164static ggml_et_cpu_compare_config concat_cpu_compare_config = {165 /* .enabled = */ false,166 /* .use_cpu_result = */ false,167 /* .log_differences = */ true,168 /* .tolerance = */ 1e-6f,169 /* .max_log_elements = */ 4096170};171 172static ggml_et_cpu_compare_config cumsum_cpu_compare_config = {173 /* .enabled = */ false,174 /* .use_cpu_result = */ false,175 /* .log_differences = */ true,176 /* .tolerance = */ 1e-6f,177 /* .max_log_elements = */ 4096178};179 180static ggml_et_cpu_compare_config repeat_cpu_compare_config = {181 /* .enabled = */ false,182 /* .use_cpu_result = */ false,183 /* .log_differences = */ true,184 /* .tolerance = */ 1e-6f,185 /* .max_log_elements = */ 4096186};187 188static ggml_et_cpu_compare_config ssm_conv_cpu_compare_config = {189 /* .enabled = */ false,190 /* .use_cpu_result = */ false,191 /* .log_differences = */ true,192 /* .tolerance = */ 1e-6f,193 /* .max_log_elements = */ 4096194};195 196static ggml_et_cpu_compare_config rwkv_wkv6_cpu_compare_config = {197 /* .enabled = */ false,198 /* .use_cpu_result = */ false,199 /* .log_differences = */ true,200 /* .tolerance = */ 1e-4f,201 /* .max_log_elements = */ 4096202};203 204static ggml_et_cpu_compare_config rwkv_wkv7_cpu_compare_config = {205 /* .enabled = */ false,206 /* .use_cpu_result = */ false,207 /* .log_differences = */ true,208 /* .tolerance = */ 1e-4f,209 /* .max_log_elements = */ 4096210};211 212static ggml_et_cpu_compare_config set_rows_cpu_compare_config = {213 /* .enabled = */ false,214 /* .use_cpu_result = */ false,215 /* .log_differences = */ true,216 /* .tolerance = */ 1e-6f,217 /* .max_log_elements = */ 2048218};219 220bool ggml_et_op_rms_norm_mul(ggml_backend_et_device_context * dev_ctx,221 const ggml_tensor * rms_norm_node,222 const ggml_tensor * mul_node) {223 ET_PERF_START();224 225 if (!dev_ctx || !rms_norm_node || !mul_node) {226 GGML_LOG_ERROR("ET: Invalid parameters for fused RMS_NORM_MUL operation\n");227 return false;228 }229 230 if (!rms_norm_node->src[0]) {231 GGML_LOG_ERROR("ET: Fused RMS_NORM_MUL missing required input\n");232 return false;233 }234 235 // Extract weights: the MUL operand that isn't the rms_norm output236 const ggml_tensor * weights = (mul_node->src[0] == rms_norm_node) ? mul_node->src[1] : mul_node->src[0];237 238 if (!weights) {239 GGML_LOG_ERROR("ET: Fused RMS_NORM_MUL missing weights tensor\n");240 return false;241 }242 243 float eps;244 memcpy(&eps, rms_norm_node->op_params, sizeof(float));245 246 ggml_et_rms_norm_mul_params params;247 params.src0 = *rms_norm_node->src[0]; // input to normalize248 params.src1 = *weights; // normalization weights249 params.dst = *mul_node; // final output250 params.eps = eps;251 252 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "rms_norm_mul_f32", ¶ms, sizeof(params), 0xFFFFFFFF);253 254 ET_PERF_END_EXT("RMS_NORM_MUL", "rms_norm_mul_f32", mul_node, "eps=%.6f", (double) eps);255 return kernel_result;256}257 258bool ggml_et_op_scale(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {259 ET_PERF_START();260 261 if (!dev_ctx || !node) {262 GGML_LOG_ERROR("ET: Invalid parameters for SCALE operation\n");263 return false;264 }265 266 if (!node->src[0]) {267 GGML_LOG_ERROR("ET: SCALE operation missing required input\n");268 return false;269 }270 271 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {272 GGML_LOG_ERROR("ET: SCALE operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),273 ggml_type_name(node->src[0]->type));274 return false;275 }276 277 float scale, bias;278 memcpy(&scale, (const float *) node->op_params + 0, sizeof(float));279 memcpy(&bias, (const float *) node->op_params + 1, sizeof(float));280 281 ggml_et_scale_params params;282 params.src0 = *node->src[0];283 params.dst = *node;284 params.scale = scale;285 params.bias = bias;286 287 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "scale_f32", ¶ms, sizeof(params), 0xFFFFFFFF);288 289 ET_PERF_END_EXT("SCALE", "scale_f32", node, "scale=%.6f|bias=%.6f", (double) scale, (double) bias);290 return kernel_result;291}292 293bool ggml_et_op_sqr(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {294 ET_PERF_START();295 296 if (!dev_ctx || !node) {297 GGML_LOG_ERROR("ET: Invalid parameters for SQR operation\n");298 return false;299 }300 301 if (!node->src[0]) {302 GGML_LOG_ERROR("ET: SQR operation missing required input\n");303 return false;304 }305 306 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {307 GGML_LOG_ERROR("ET: SQR operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),308 ggml_type_name(node->src[0]->type));309 return false;310 }311 312 ggml_et_sqr_params params;313 params.src0 = *node->src[0]; // F32 input tensor314 params.dst = *node; // F32 output tensor315 316 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)317 ggml_et_cpu_compare_ctx cpu_cmp_ctx;318 bool cpu_comparison_active = false;319 if (sqr_cpu_compare_config.enabled) {320 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_SQR)) {321 cpu_comparison_active = true;322 } else {323 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for SQR operation\n");324 }325 }326 327 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "sqr_f32", ¶ms, sizeof(params), 0xFFFFFFFF);328 329 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)330 if (cpu_comparison_active) {331 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &sqr_cpu_compare_config)) {332 GGML_LOG_WARN("ET: CPU comparison failed for SQR operation\n");333 }334 ggml_et_cpu_compare_free(&cpu_cmp_ctx);335 }336 337 ET_PERF_END("SQR", "sqr_f32", node);338 return kernel_result;339}340 341bool ggml_et_op_sum_rows(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {342 ET_PERF_START();343 344 if (!dev_ctx || !node) {345 GGML_LOG_ERROR("ET: Invalid parameters for SUM_ROWS operation\n");346 return false;347 }348 349 if (!node->src[0]) {350 GGML_LOG_ERROR("ET: SUM_ROWS operation missing required input\n");351 return false;352 }353 354 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {355 GGML_LOG_ERROR("ET: SUM_ROWS operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),356 ggml_type_name(node->src[0]->type));357 return false;358 }359 360 ggml_et_sum_rows_params params;361 params.src0 = *node->src[0];362 params.dst = *node;363 364 // Phase 1: Initialize CPU comparison context365 ggml_et_cpu_compare_ctx cpu_cmp_ctx;366 bool cpu_comparison_active = false;367 if (sum_rows_cpu_compare_config.enabled) {368 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_SUM_ROWS)) {369 cpu_comparison_active = true;370 } else {371 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for SUM_ROWS operation\n");372 }373 }374 375 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "sum_rows_f32", ¶ms, sizeof(params), 0xFFFFFFFF);376 377 // Phase 2: Execute CPU computation and compare378 if (cpu_comparison_active) {379 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &sum_rows_cpu_compare_config)) {380 GGML_LOG_WARN("ET: CPU comparison failed for SUM_ROWS operation\n");381 }382 ggml_et_cpu_compare_free(&cpu_cmp_ctx);383 }384 385 ET_PERF_END("SUM_ROWS", "sum_rows_f32", node);386 return kernel_result;387}388 389bool ggml_et_op_mean(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {390 ET_PERF_START();391 392 if (!dev_ctx || !node) {393 GGML_LOG_ERROR("ET: Invalid parameters for MEAN operation\n");394 return false;395 }396 397 if (!node->src[0]) {398 GGML_LOG_ERROR("ET: MEAN operation missing required input\n");399 return false;400 }401 402 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {403 GGML_LOG_ERROR("ET: MEAN operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),404 ggml_type_name(node->src[0]->type));405 return false;406 }407 408 ggml_et_mean_params params;409 params.src0 = *node->src[0];410 params.dst = *node;411 412 ggml_et_cpu_compare_ctx cpu_cmp_ctx;413 bool cpu_comparison_active = false;414 if (mean_cpu_compare_config.enabled) {415 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_MEAN)) {416 cpu_comparison_active = true;417 } else {418 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for MEAN operation\n");419 }420 }421 422 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "mean_f32", ¶ms, sizeof(params), 0xFFFFFFFF);423 424 if (cpu_comparison_active) {425 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &mean_cpu_compare_config)) {426 GGML_LOG_WARN("ET: CPU comparison failed for MEAN operation\n");427 }428 ggml_et_cpu_compare_free(&cpu_cmp_ctx);429 }430 431 ET_PERF_END("MEAN", "mean_f32", node);432 return kernel_result;433}434 435bool ggml_et_op_clamp(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {436 ET_PERF_START();437 438 if (!dev_ctx || !node) {439 GGML_LOG_ERROR("ET: Invalid parameters for CLAMP operation\n");440 return false;441 }442 443 if (!node->src[0]) {444 GGML_LOG_ERROR("ET: CLAMP operation missing required input\n");445 return false;446 }447 448 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {449 GGML_LOG_ERROR("ET: CLAMP operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),450 ggml_type_name(node->src[0]->type));451 return false;452 }453 454 ggml_et_clamp_params params;455 params.src0 = *node->src[0];456 params.dst = *node;457 // op_params layout per ggml.c::ggml_clamp: { min, max } as floats458 memcpy(¶ms.min_val, (const float *) node->op_params + 0, sizeof(float));459 memcpy(¶ms.max_val, (const float *) node->op_params + 1, sizeof(float));460 461 ggml_et_cpu_compare_ctx cpu_cmp_ctx;462 bool cpu_comparison_active = false;463 if (clamp_cpu_compare_config.enabled) {464 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_CLAMP)) {465 cpu_comparison_active = true;466 } else {467 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for CLAMP operation\n");468 }469 }470 471 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "clamp_f32", ¶ms, sizeof(params), 0xFFFFFFFF);472 473 if (cpu_comparison_active) {474 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &clamp_cpu_compare_config)) {475 GGML_LOG_WARN("ET: CPU comparison failed for CLAMP operation\n");476 }477 ggml_et_cpu_compare_free(&cpu_cmp_ctx);478 }479 480 ET_PERF_END("CLAMP", "clamp_f32", node);481 return kernel_result;482}483 484bool ggml_et_op_unary(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {485 ET_PERF_START();486 487 if (!dev_ctx || !node) {488 GGML_LOG_ERROR("ET: Invalid parameters for UNARY operation\n");489 return false;490 }491 492 if (!node->src[0]) {493 GGML_LOG_ERROR("ET: UNARY operation missing required input\n");494 return false;495 }496 497 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {498 GGML_LOG_ERROR("ET: UNARY operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),499 ggml_type_name(node->src[0]->type));500 return false;501 }502 503 const ggml_unary_op uop = ggml_get_unary_op(node);504 const char * op_name = ggml_unary_op_name(uop);505 506 ggml_et_unary_params params;507 params.src0 = *node->src[0]; // F32 input tensor508 params.dst = *node; // F32 output tensor509 params.unary_op = (int32_t) uop;510 511 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)512 ggml_et_cpu_compare_ctx cpu_cmp_ctx;513 bool cpu_comparison_active = false;514 if (unary_cpu_compare_config.enabled) {515 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_UNARY)) {516 cpu_comparison_active = true;517 } else {518 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for UNARY/%s operation\n", op_name);519 }520 }521 522 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "unary_f32", ¶ms, sizeof(params), 0xFFFFFFFF);523 524 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)525 if (cpu_comparison_active) {526 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &unary_cpu_compare_config)) {527 GGML_LOG_WARN("ET: CPU comparison failed for UNARY/%s operation\n", op_name);528 }529 ggml_et_cpu_compare_free(&cpu_cmp_ctx);530 }531 532 ET_PERF_END_EXT("UNARY", "unary_f32", node, "op=%s", op_name);533 return kernel_result;534}535 536bool ggml_et_op_mul(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {537 // Delegate to generic element map operation538 return ggml_et_op_elmap(dev_ctx, node);539}540 541bool ggml_et_op_add(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {542 // Delegate to generic element map operation543 return ggml_et_op_elmap(dev_ctx, node);544}545 546bool ggml_et_op_sub(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {547 // Delegate to generic element map operation548 return ggml_et_op_elmap(dev_ctx, node);549}550 551bool ggml_et_op_elmap(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {552 ET_PERF_START();553 554 if (!dev_ctx || !node) {555 GGML_LOG_ERROR("ET: Invalid parameters for element map operation\n");556 return false;557 }558 559 if (!node->src[0] || !node->src[1]) {560 GGML_LOG_ERROR("ET: Element map operation missing required inputs\n");561 return false;562 }563 564 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32 || node->src[1]->type != GGML_TYPE_F32) {565 GGML_LOG_ERROR("ET: Element map operation with unsupported types: dst=%s src0=%s src1=%s\n",566 ggml_type_name(node->type), ggml_type_name(node->src[0]->type),567 ggml_type_name(node->src[1]->type));568 return false;569 }570 571 const char * op_name = ggml_op_name(node->op);572 573 ggml_et_elmap_params params;574 params.src0 = *node->src[0];575 params.src1 = *node->src[1];576 params.dst = *node; // F32 output tensor (op type stored in dst.op)577 578 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)579 ggml_et_cpu_compare_ctx cpu_cmp_ctx;580 bool cpu_comparison_active = false;581 if (elmap_cpu_compare_config.enabled) {582 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, node->op)) {583 cpu_comparison_active = true;584 } else {585 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for %s operation\n", op_name);586 }587 }588 589 // fprintf(stderr, "ET: el_map s0 [%ld, %ld, %ld, %ld] s1 [%ld, %ld, %ld, %ld]\n",590 // node->src[0]->ne[0], node->src[0]->ne[1], node->src[0]->ne[2], node->src[0]->ne[3],591 // node->src[1]->ne[0], node->src[1]->ne[1], node->src[1]->ne[2], node->src[1]->ne[3]);592 593 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "el_map_f32", ¶ms, sizeof(params), 0xFFFFFFFF);594 595 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)596 if (cpu_comparison_active) {597 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &elmap_cpu_compare_config)) {598 GGML_LOG_WARN("ET: CPU comparison failed for %s operation\n", op_name);599 }600 ggml_et_cpu_compare_free(&cpu_cmp_ctx);601 }602 603 ET_PERF_END(op_name, "el_map_f32", node);604 return kernel_result;605}606 607bool ggml_et_op_glu(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {608 ET_PERF_START();609 610 // Validate inputs611 if (!dev_ctx || !node) {612 GGML_LOG_ERROR("ET: Invalid parameters for GLU operation\n");613 return false;614 }615 616 if (!node->src[0]) {617 GGML_LOG_ERROR("ET: GLU operation missing required input\n");618 return false;619 }620 621 const bool is_split_mode = node->src[1] != nullptr;622 623 // Only support F32 (as validated by supports_op)624 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32 ||625 (is_split_mode && node->src[1]->type != GGML_TYPE_F32)) {626 return false;627 }628 629 // Extract GLU operation parameters from op_params630 int32_t glu_op_type = ggml_get_op_params_i32(node, 0); // GLU variant (REGLU, GEGLU, SWIGLU, etc.)631 int32_t swapped = ggml_get_op_params_i32(node, 1); // Whether gate/value are swapped632 633 // Supported variants634 switch (glu_op_type) {635 case GGML_GLU_OP_REGLU:636 case GGML_GLU_OP_GEGLU:637 case GGML_GLU_OP_SWIGLU:638 case GGML_GLU_OP_SWIGLU_OAI:639 case GGML_GLU_OP_GEGLU_ERF:640 case GGML_GLU_OP_GEGLU_QUICK:641 break;642 default:643 GGML_LOG_ERROR("ET: GLU operation with unsupported variant: %s\n",644 ggml_glu_op_name((ggml_glu_op) glu_op_type));645 return false;646 }647 648 // Get GLU operation name for logging649 const char * glu_op_name = ggml_glu_op_name((ggml_glu_op) glu_op_type);650 651 // Pack parameters. Single-tensor mode is encoded by zeroing src1.652 ggml_et_glu_params params = {};653 params.src0 = *node->src[0];654 if (is_split_mode) {655 params.src1 = *node->src[1];656 }657 params.dst = *node;658 params.glu_op_type = glu_op_type;659 params.swapped = swapped;660 params.alpha = 0.0f;661 params.limit = 0.0f;662 if (glu_op_type == GGML_GLU_OP_SWIGLU_OAI) {663 params.alpha = ggml_get_op_params_f32(node, 2);664 params.limit = ggml_get_op_params_f32(node, 3);665 }666 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)667 ggml_et_cpu_compare_ctx cpu_cmp_ctx;668 bool cpu_comparison_active = false;669 if (glu_cpu_compare_config.enabled) {670 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_GLU)) {671 cpu_comparison_active = true;672 } else {673 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for %s operation\n", glu_op_name);674 }675 }676 677 // Launch ET kernel678 bool kernel_result = ggml_et_launch_kernel(dev_ctx, "glu_f32", ¶ms, sizeof(params), 0xFFFFFFFF);679 680 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)681 if (cpu_comparison_active) {682 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &glu_cpu_compare_config)) {683 GGML_LOG_WARN("ET: CPU comparison failed for %s operation\n", glu_op_name);684 }685 ggml_et_cpu_compare_free(&cpu_cmp_ctx);686 }687 688 ET_PERF_END("GLU", "glu_f32", node);689 return kernel_result;690}691 692bool ggml_et_op_mul_mat(ggml_backend_et_device_context * dev_ctx,693 const ggml_tensor * node,694 const ggml_tensor * add_node) {695 ET_PERF_START();696 697 if (!dev_ctx || !node) {698 GGML_LOG_ERROR("ET: Invalid parameters for MUL_MAT operation\n");699 return false;700 }701 702 if (!node->src[0] || !node->src[1]) {703 GGML_LOG_ERROR("ET: MUL_MAT operation missing required inputs\n");704 return false;705 }706 707 // Fused MM+ADD: when add_node is non-NULL the caller has already validated708 // (Q8_0 weights, F32 acts, exact-shape ADD with stride parity to dst) via709 // ggml_et_can_fuse({MUL_MAT, ADD}). The kernel writes dst = mm + bias and710 // the ADD's output replaces MM's as the actual dst.711 const ggml_tensor * fused_dst = add_node ? add_node : node;712 const ggml_tensor * bias_tensor = nullptr;713 if (add_node) {714 bias_tensor = (add_node->src[0] == node) ? add_node->src[1] : add_node->src[0];715 }716 717 const char * kernel_name;718 const char * src0_type_name;719 720 if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q4_0 && node->src[1]->type == GGML_TYPE_F32 &&721 node->src[1]->ne[1] >= 53 && // N >= 53722 node->src[0]->ne[1] % 16 == 0 && // M % TILE_M723 node->src[0]->ne[0] % 32 == 0) { // K % BLOCK_K (Q4_0 block)724 725 // Matrix engine for N >= 53; partial N (via n_cur-1) and errata padding are handled in-kernel.726 kernel_name = "mul_mat_Q4_0_matrix_engine";727 src0_type_name = "Q4_0";728 729 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q4_0 &&730 node->src[1]->type == GGML_TYPE_F32) {731 kernel_name = "mul_mat_Q4_0"; // N < 53, or M % 16 != 0 or K % 32 != 0732 src0_type_name = "Q4_0";733 734 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q8_0 &&735 node->src[1]->type == GGML_TYPE_F32) {736 kernel_name = "mul_mat_Q8_0";737 src0_type_name = "Q8_0";738 739 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F16 &&740 node->src[1]->type == GGML_TYPE_F16 && node->ne[0] % 16 == 0 && node->src[0]->ne[0] % 16 == 0 &&741 node->src[0]->ne[1] % 16 == 0 && node->src[1]->ne[0] != 1) {742 kernel_name = "mul_mat_f16_matrix_engine";743 src0_type_name = "F16";744 745 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F16 &&746 (node->src[1]->type == GGML_TYPE_F16 || node->src[1]->type == GGML_TYPE_F32)) {747 kernel_name = "mul_mat_f16";748 src0_type_name = "F16";749 750 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32 &&751 node->src[1]->type == GGML_TYPE_F32 && node->ne[0] % 16 == 0 && node->src[0]->ne[0] % 16 == 0 &&752 node->src[0]->ne[1] % 16 == 0 && node->src[1]->ne[0] != 1) { // GEMV is faster with the generic path753 754 kernel_name = "mul_mat_f32_matrix_engine";755 src0_type_name = "F32";756 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32 &&757 (node->src[1]->type == GGML_TYPE_F16 || node->src[1]->type == GGML_TYPE_F32)) {758 kernel_name = "mul_mat_f32";759 src0_type_name = "F32";760 } else {761 GGML_LOG_ERROR("ET: MUL_MAT operation with unsupported types: dst=%s src0=%s src1=%s\n",762 ggml_type_name(node->type), ggml_type_name(node->src[0]->type),763 ggml_type_name(node->src[1]->type));764 return false;765 }766 767 ggml_et_binary_params params;768 params.src0 = *node->src[0]; // weight matrix769 params.src1 = *node->src[1]; // activation matrix770 params.dst = *fused_dst; // output (= add_node when fused, else node)771 772 ggml_et_cpu_compare_ctx cpu_cmp_ctx;773 bool cpu_comparison_active = false;774 if (mul_mat_cpu_compare_config.enabled) {775 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, fused_dst, GGML_OP_MUL_MAT)) {776 cpu_comparison_active = true;777 } else {778 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for MUL_MAT operation\n");779 }780 }781 782 bool kernel_result;783 if (node->src[0]->type == GGML_TYPE_Q8_0) {784 // Q8_0 kernel always takes the extended struct. bias.data is non-NULL785 // only on the fused path; otherwise the kernel skips the add entirely.786 ggml_et_mm_q8_params q8_params = {};787 q8_params.src0 = params.src0;788 q8_params.src1 = params.src1;789 q8_params.dst = params.dst;790 if (bias_tensor) {791 q8_params.bias = *bias_tensor;792 }793 kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, &q8_params, sizeof(q8_params), 0xFFFFFFFF);794 } else {795 // Non-Q8 MM kernels don't yet support fused-add; the graph fuse check796 // already rejects non-Q8 pairs, so add_node is always nullptr here.797 kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, ¶ms, sizeof(params), 0xFFFFFFFF);798 }799 800 // printf("Tensor error:");801 // if (params.src0.data != NULL)802 // {803 // printf("Ptr OK\n");804 // printf("node->data ptr = %p\n", node->data);805 // // if (once < 100){806 // // // uint64_t * host_data = (uint64_t *) node->data;807 // // // printf("Tensor error: %lu\n", host_data[0]);808 809 // // // printf("Tensor error:");810 // // once++;811 // // }812 // }813 814 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)815 if (cpu_comparison_active) {816 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, fused_dst, &mul_mat_cpu_compare_config)) {817 GGML_LOG_WARN("ET: CPU comparison failed for MUL_MAT operation\n");818 }819 ggml_et_cpu_compare_free(&cpu_cmp_ctx);820 }821 822 {823 // Calculate actual FLOPs including batch/sequence dimensions824 // dst shape: [M, N, ne2, ne3] where M=ne[1], N=ne[0]825 int64_t m = node->ne[1];826 int64_t n = node->ne[0];827 int64_t k = node->src[0]->ne[0];828 int64_t ne2 = node->ne[2];829 int64_t ne3 = node->ne[3];830 831 // Total FLOPs = (batch_size) * M * N * (2*K - 1)832 // Each MxN matrix-matrix multiply does M*N*(2*K-1) FLOPs833 // Broadcasting is handled by repeating computation, so count actual operations834 int64_t batch_size = ne2 * ne3;835 int64_t total_flops = batch_size * m * n * (2 * k - 1);836 837 char kernel_variant[64];838 snprintf(kernel_variant, sizeof(kernel_variant), "%s_%sx%s", kernel_name, src0_type_name,839 ggml_type_name(node->src[1]->type));840 ET_PERF_END_EXT("MUL_MAT", kernel_variant, node, "flops=%" PRId64, total_flops);841 }842 return kernel_result;843}844 845bool ggml_et_op_mul_mat_id(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {846 ET_PERF_START();847 if (!dev_ctx || !node) {848 GGML_LOG_ERROR("ET: Invalid parameters for MUL_MAT_ID operation\n");849 return false;850 }851 852 if (!node->src[0] || !node->src[1] || !node->src[2]) {853 GGML_LOG_ERROR("ET: MUL_MAT_ID operation missing required inputs\n");854 return false;855 }856 857 const char * kernel_name;858 const char * src0_type_name;859 860 // Support Q8_0/Q4_0/F16/F32 x F32 -> F32 matrix multiplication with expert selection861 if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q8_0 && node->src[1]->type == GGML_TYPE_F32 &&862 node->src[2]->type == GGML_TYPE_I32) {863 kernel_name = "mul_mat_id_Q8_0";864 src0_type_name = "Q8_0";865 866 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q4_0 &&867 node->src[1]->type == GGML_TYPE_F32 && node->src[2]->type == GGML_TYPE_I32) {868 kernel_name = "mul_mat_id_Q4_0";869 src0_type_name = "Q4_0";870 871 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F16 &&872 node->src[1]->type == GGML_TYPE_F32 && node->src[2]->type == GGML_TYPE_I32) {873 kernel_name = "mul_mat_id_f32";874 src0_type_name = "F16";875 876 } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32 &&877 node->src[1]->type == GGML_TYPE_F32 && node->src[2]->type == GGML_TYPE_I32) {878 kernel_name = "mul_mat_id_f32";879 src0_type_name = "F32";880 881 } else {882 GGML_LOG_ERROR("ET: MUL_MAT_ID operation with unsupported types: dst=%s src0=%s src1=%s src2=%s\n",883 ggml_type_name(node->type), ggml_type_name(node->src[0]->type),884 ggml_type_name(node->src[1]->type), ggml_type_name(node->src[2]->type));885 return false;886 }887 888 // Pack parameters - copy full tensor structures889 ggml_et_mul_mat_id_params params;890 params.src0 = *node->src[0]; // Expert weight matrices (Q8_0/F16/F32)891 params.src1 = *node->src[1]; // Activation matrix (F32)892 params.src2 = *node->src[2]; // Expert indices (I32)893 params.dst = *node; // Output matrix (F32)894 895 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)896 ggml_et_cpu_compare_ctx cpu_cmp_ctx;897 bool cpu_comparison_active = false;898 if (mul_mat_id_cpu_compare_config.enabled) {899 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_MUL_MAT_ID)) {900 cpu_comparison_active = true;901 } else {902 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for MUL_MAT_ID operation\n");903 }904 }905 906 // Launch ET kernel907 bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, ¶ms, sizeof(params), 0xFFFFFFFF);908 909 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)910 if (cpu_comparison_active) {911 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &mul_mat_id_cpu_compare_config)) {912 GGML_LOG_WARN("ET: CPU comparison failed for MUL_MAT_ID operation\n");913 }914 ggml_et_cpu_compare_free(&cpu_cmp_ctx);915 }916 917 // Calculate FLOPs (approximate - similar to MUL_MAT but with expert routing overhead)918 // Each expert computation is similar to a MUL_MAT, but we only compute for selected experts919 int64_t K = node->src[0]->ne[0];920 int64_t M = node->src[0]->ne[1];921 int64_t n_expert_used = node->src[2]->ne[0];922 int64_t batch = node->src[2]->ne[1];923 924 int64_t total_flops = batch * n_expert_used * M * (2 * K - 1);925 926 char kernel_variant[64];927 snprintf(kernel_variant, sizeof(kernel_variant), "%s_%sx%s", kernel_name, src0_type_name,928 ggml_type_name(node->src[1]->type));929 ET_PERF_END_EXT("MUL_MAT_ID", kernel_variant, node, "flops=%" PRId64 "|n_expert=%lld|n_expert_used=%lld",930 total_flops, (long long) node->src[0]->ne[2], (long long) n_expert_used);931 932 return kernel_result;933}934 935bool ggml_et_op_rope(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {936 ET_PERF_START();937 938 if (!dev_ctx || !node) {939 GGML_LOG_ERROR("ET: Invalid parameters for ROPE operation\n");940 return false;941 }942 943 if (!node->src[0] || !node->src[1]) {944 GGML_LOG_ERROR("ET: ROPE operation missing required inputs\n");945 return false;946 }947 948 const char * kernel_name;949 950 if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32 && node->src[1]->type == GGML_TYPE_I32) {951 kernel_name = "rope_f32";952 } else {953 return false;954 }955 956 // Pack parameters - copy full tensor structures and op_params957 ggml_et_rope_params params;958 params.src0 = *node->src[0]; // F32 input tensor959 params.src1 = *node->src[1]; // I32 position tensor960 if (node->src[2]) {961 params.src2 = *node->src[2]; // F32 frequency factors (optional)962 } else {963 memset(¶ms.src2, 0, sizeof(params.src2)); // Zero if not provided964 }965 params.dst = *node; // F32 output tensor966 967 params.rope_params.n_past = ((const int32_t *) node->op_params)[0];968 params.rope_params.n_dims = ((const int32_t *) node->op_params)[1];969 params.rope_params.mode = ((const int32_t *) node->op_params)[2];970 params.rope_params.n_ctx = ((const int32_t *) node->op_params)[3];971 params.rope_params.n_ctx_orig = ((const int32_t *) node->op_params)[4];972 memcpy(¶ms.rope_params.freq_base, (const int32_t *) node->op_params + 5, sizeof(float));973 memcpy(¶ms.rope_params.freq_scale, (const int32_t *) node->op_params + 6, sizeof(float));974 memcpy(¶ms.rope_params.ext_factor, (const int32_t *) node->op_params + 7, sizeof(float));975 memcpy(¶ms.rope_params.attn_factor, (const int32_t *) node->op_params + 8, sizeof(float));976 memcpy(¶ms.rope_params.beta_fast, (const int32_t *) node->op_params + 9, sizeof(float));977 memcpy(¶ms.rope_params.beta_slow, (const int32_t *) node->op_params + 10, sizeof(float));978 if (params.rope_params.mode & GGML_ROPE_TYPE_MROPE) {979 memcpy(params.rope_params.sections, (const int32_t *) node->op_params + 11, sizeof(int32_t) * 4);980 } else {981 memset(params.rope_params.sections, 0, sizeof(params.rope_params.sections));982 }983 984 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)985 ggml_et_cpu_compare_ctx cpu_cmp_ctx;986 bool cpu_comparison_active = false;987 if (rope_cpu_compare_config.enabled) {988 GGML_LOG_DEBUG("ET: Initializing CPU comparison for ROPE operation\n");989 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_ROPE)) {990 cpu_comparison_active = true;991 } else {992 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for ROPE operation\n");993 }994 }995 996 bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, ¶ms, sizeof(params), 0xFFFFFFFF);997 998 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)999 if (cpu_comparison_active) {1000 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &rope_cpu_compare_config)) {1001 GGML_LOG_WARN("ET: CPU comparison failed for ROPE operation\n");1002 }1003 ggml_et_cpu_compare_free(&cpu_cmp_ctx);1004 }1005 1006 ET_PERF_END_EXT("ROPE", kernel_name, node, "mode=0x%x|n_dims=%d|freq_base=%.2f|freq_scale=%.2f",1007 params.rope_params.mode, params.rope_params.n_dims, (double) params.rope_params.freq_base,1008 (double) params.rope_params.freq_scale);1009 return kernel_result;1010}1011 1012bool ggml_et_op_rms_norm(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {1013 ET_PERF_START();1014 1015 if (!dev_ctx || !node) {1016 GGML_LOG_ERROR("ET: Invalid parameters for RMS_NORM operation\n");1017 return false;1018 }1019 1020 if (!node->src[0]) {1021 GGML_LOG_ERROR("ET: RMS_NORM operation missing required input\n");1022 return false;1023 }1024 1025 const char * kernel_name;1026 1027 if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32) {1028 kernel_name = "rms_norm_f32";1029 1030 } else {1031 GGML_LOG_ERROR("ET: RMS_NORM operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),1032 ggml_type_name(node->src[0]->type));1033 return false;1034 }1035 1036 float eps;1037 memcpy(&eps, node->op_params, sizeof(float));1038 1039 ggml_et_rms_norm_params params;1040 params.src0 = *node->src[0]; // F32 input tensor1041 params.dst = *node; // F32 output tensor1042 params.eps = eps; // Epsilon parameter for numerical stability1043 1044 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)1045 ggml_et_cpu_compare_ctx cpu_cmp_ctx;1046 bool cpu_comparison_active = false;1047 if (rms_norm_cpu_compare_config.enabled) {1048 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_RMS_NORM)) {1049 cpu_comparison_active = true;1050 } else {1051 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for RMS_NORM operation\n");1052 }1053 }1054 1055 bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, ¶ms, sizeof(params), 0xFFFFFFFF);1056 1057 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)1058 if (cpu_comparison_active) {1059 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &rms_norm_cpu_compare_config)) {1060 GGML_LOG_WARN("ET: CPU comparison failed for RMS_NORM operation\n");1061 }1062 ggml_et_cpu_compare_free(&cpu_cmp_ctx);1063 }1064 1065 ET_PERF_END_EXT("RMS_NORM", kernel_name, node, "eps=%.6f", (double) eps);1066 return kernel_result;1067}1068 1069bool ggml_et_op_norm(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {1070 ET_PERF_START();1071 1072 if (!dev_ctx || !node) {1073 GGML_LOG_ERROR("ET: Invalid parameters for NORM operation\n");1074 return false;1075 }1076 1077 if (!node->src[0]) {1078 GGML_LOG_ERROR("ET: NORM operation missing required input\n");1079 return false;1080 }1081 1082 const char * kernel_name;1083 1084 if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32) {1085 kernel_name = "norm_f32";1086 1087 } else {1088 GGML_LOG_ERROR("ET: NORM operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),1089 ggml_type_name(node->src[0]->type));1090 return false;1091 }1092 1093 float eps;1094 memcpy(&eps, node->op_params, sizeof(float));1095 1096 ggml_et_norm_params params;1097 params.src0 = *node->src[0]; // F32 input tensor1098 params.dst = *node; // F32 output tensor1099 params.eps = eps; // Epsilon parameter for numerical stability1100 1101 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)1102 ggml_et_cpu_compare_ctx cpu_cmp_ctx;1103 bool cpu_comparison_active = false;1104 if (norm_cpu_compare_config.enabled) {1105 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_NORM)) {1106 cpu_comparison_active = true;1107 } else {1108 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for NORM operation\n");1109 }1110 }1111 1112 bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, ¶ms, sizeof(params), 0xFFFFFFFF);1113 1114 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)1115 if (cpu_comparison_active) {1116 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &norm_cpu_compare_config)) {1117 GGML_LOG_WARN("ET: CPU comparison failed for NORM operation\n");1118 }1119 ggml_et_cpu_compare_free(&cpu_cmp_ctx);1120 }1121 1122 ET_PERF_END_EXT("NORM", kernel_name, node, "eps=%.6f", (double) eps);1123 return kernel_result;1124}1125 1126bool ggml_et_op_l2_norm(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {1127 ET_PERF_START();1128 1129 if (!dev_ctx || !node) {1130 GGML_LOG_ERROR("ET: Invalid parameters for L2_NORM operation\n");1131 return false;1132 }1133 1134 if (!node->src[0]) {1135 GGML_LOG_ERROR("ET: L2_NORM operation missing required input\n");1136 return false;1137 }1138 1139 const char * kernel_name;1140 1141 if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32) {1142 kernel_name = "l2_norm_f32";1143 1144 } else {1145 GGML_LOG_ERROR("ET: L2_NORM operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),1146 ggml_type_name(node->src[0]->type));1147 return false;1148 }1149 1150 float eps;1151 memcpy(&eps, node->op_params, sizeof(float));1152 1153 ggml_et_l2_norm_params params;1154 params.src0 = *node->src[0]; // F32 input tensor1155 params.dst = *node; // F32 output tensor1156 params.eps = eps; // Epsilon parameter for numerical stability1157 1158 // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)1159 ggml_et_cpu_compare_ctx cpu_cmp_ctx;1160 bool cpu_comparison_active = false;1161 if (l2_norm_cpu_compare_config.enabled) {1162 if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_L2_NORM)) {1163 cpu_comparison_active = true;1164 } else {1165 GGML_LOG_WARN("ET: Failed to initialize CPU comparison for L2_NORM operation\n");1166 }1167 }1168 1169 bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, ¶ms, sizeof(params), 0xFFFFFFFF);1170 1171 // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)1172 if (cpu_comparison_active) {1173 if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &l2_norm_cpu_compare_config)) {1174 GGML_LOG_WARN("ET: CPU comparison failed for L2_NORM operation\n");1175 }1176 ggml_et_cpu_compare_free(&cpu_cmp_ctx);1177 }1178 1179 ET_PERF_END_EXT("L2_NORM", kernel_name, node, "eps=%.6f", (double) eps);1180 return kernel_result;1181}1182 1183bool ggml_et_op_group_norm(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {1184 ET_PERF_START();1185 1186 if (!dev_ctx || !node) {1187 GGML_LOG_ERROR("ET: Invalid parameters for GROUP_NORM operation\n");1188 return false;1189 }1190 1191 if (!node->src[0]) {1192 GGML_LOG_ERROR("ET: GROUP_NORM operation missing required input\n");1193 return false;1194 }1195 1196 if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {1197 GGML_LOG_ERROR("ET: GROUP_NORM operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),1198 ggml_type_name(node->src[0]->type));1199 return false;1200 }