Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
ggml-et-ops.cpp2581 linesDownload Raw Back to ggml-et
1#include "ggml-et-ops.h"2 3#include "ggml-et-cpu-compare.h"4#include "ggml-et-kernels.h"5#include "ggml-impl.h"6 7#include <stdio.h>8 9#include <cstdint>10 11// CPU comparison configuration - can be enabled for debugging12static ggml_et_cpu_compare_config rope_cpu_compare_config = {13    /* .enabled = */ false,14    /* .use_cpu_result = */ false,  // Replace ET result with CPU result15    /* .log_differences = */ true,16    /* .tolerance = */ 1e-5f,17    /* .max_log_elements = */ 409618};19 20static ggml_et_cpu_compare_config rms_norm_cpu_compare_config = {21    /* .enabled = */ false,22    /* .use_cpu_result = */ false,23    /* .log_differences = */ true,24    /* .tolerance = */ 1e-5f,25    /* .max_log_elements = */ 409626};27 28static ggml_et_cpu_compare_config norm_cpu_compare_config = {29    /* .enabled = */ false,30    /* .use_cpu_result = */ false,31    /* .log_differences = */ true,32    /* .tolerance = */ 1e-5f,33    /* .max_log_elements = */ 409634};35 36static ggml_et_cpu_compare_config l2_norm_cpu_compare_config = {37    /* .enabled = */ false,38    /* .use_cpu_result = */ false,39    /* .log_differences = */ true,40    /* .tolerance = */ 1e-5f,41    /* .max_log_elements = */ 409642};43 44static ggml_et_cpu_compare_config group_norm_cpu_compare_config = {45    /* .enabled = */ false,46    /* .use_cpu_result = */ false,47    /* .log_differences = */ true,48    /* .tolerance = */ 1e-5f,49    /* .max_log_elements = */ 409650};51 52static ggml_et_cpu_compare_config im2col_cpu_compare_config = {53    /* .enabled = */ false,54    /* .use_cpu_result = */ false,55    /* .log_differences = */ true,56    /* .tolerance = */ 1e-5f,57    /* .max_log_elements = */ 409658};59 60static ggml_et_cpu_compare_config unary_cpu_compare_config = {61    /* .enabled = */ false,62    /* .use_cpu_result = */ false,63    /* .log_differences = */ true,64    /* .tolerance = */ 1e-4f,65    /* .max_log_elements = */ 409666};67 68static ggml_et_cpu_compare_config sum_rows_cpu_compare_config = {69    /* .enabled = */ false,70    /* .use_cpu_result = */ false,71    /* .log_differences = */ true,72    /* .tolerance = */ 1e-5f,73    /* .max_log_elements = */ 409674};75 76static ggml_et_cpu_compare_config clamp_cpu_compare_config = {77    /* .enabled = */ false,78    /* .use_cpu_result = */ false,79    /* .log_differences = */ true,80    /* .tolerance = */ 1e-6f,81    /* .max_log_elements = */ 409682};83 84static ggml_et_cpu_compare_config mean_cpu_compare_config = {85    /* .enabled = */ false,86    /* .use_cpu_result = */ false,87    /* .log_differences = */ true,88    /* .tolerance = */ 1e-5f,89    /* .max_log_elements = */ 409690};91 92static ggml_et_cpu_compare_config sqr_cpu_compare_config = {93    /* .enabled = */ false,94    /* .use_cpu_result = */ false,95    /* .log_differences = */ true,96    /* .tolerance = */ 1e-6f,97    /* .max_log_elements = */ 409698};99 100static ggml_et_cpu_compare_config elmap_cpu_compare_config = {101    /* .enabled = */ false,102    /* .use_cpu_result = */ false,103    /* .log_differences = */ true,104    /* .tolerance = */ 1e-6f,105    /* .max_log_elements = */ 4096106};107 108static ggml_et_cpu_compare_config glu_cpu_compare_config = {109    /* .enabled = */ false,110    /* .use_cpu_result = */ false,111    /* .log_differences = */ true,112    /* .tolerance = */ 1e-5f,113    /* .max_log_elements = */ 4096114};115 116static ggml_et_cpu_compare_config mul_mat_cpu_compare_config = {117    /* .enabled = */ false,118    /* .use_cpu_result = */ false,119    /* .log_differences = */ true,120    /* .tolerance = */ 0.01,121    /* .max_log_elements = */ 4096122};123 124static ggml_et_cpu_compare_config mul_mat_id_cpu_compare_config = {125    /* .enabled = */ false,126    /* .use_cpu_result = */ false,127    /* .log_differences = */ true,128    /* .tolerance = */ 0.01,129    /* .max_log_elements = */ 4096130};131 132static ggml_et_cpu_compare_config softmax_cpu_compare_config = {133    /* .enabled = */ false,134    /* .use_cpu_result = */ false,135    /* .log_differences = */ true,136    /* .tolerance = */ 1e-5f,137    /* .max_log_elements = */ 1024138};139 140static ggml_et_cpu_compare_config get_rows_cpu_compare_config = {141    /* .enabled = */ false,142    /* .use_cpu_result = */ false,143    /* .log_differences = */ true,144    /* .tolerance = */ 1e-6f,145    /* .max_log_elements = */ 2048146};147 148static ggml_et_cpu_compare_config pad_cpu_compare_config = {149    /* .enabled = */ false,150    /* .use_cpu_result = */ false,151    /* .log_differences = */ true,152    /* .tolerance = */ 1e-6f,153    /* .max_log_elements = */ 4096154};155 156static ggml_et_cpu_compare_config cont_cpu_compare_config = {157    /* .enabled = */ false,158    /* .use_cpu_result = */ false,159    /* .log_differences = */ true,160    /* .tolerance = */ 1e-6f,161    /* .max_log_elements = */ 4096162};163 164static ggml_et_cpu_compare_config concat_cpu_compare_config = {165    /* .enabled = */ false,166    /* .use_cpu_result = */ false,167    /* .log_differences = */ true,168    /* .tolerance = */ 1e-6f,169    /* .max_log_elements = */ 4096170};171 172static ggml_et_cpu_compare_config cumsum_cpu_compare_config = {173    /* .enabled = */ false,174    /* .use_cpu_result = */ false,175    /* .log_differences = */ true,176    /* .tolerance = */ 1e-6f,177    /* .max_log_elements = */ 4096178};179 180static ggml_et_cpu_compare_config repeat_cpu_compare_config = {181    /* .enabled = */ false,182    /* .use_cpu_result = */ false,183    /* .log_differences = */ true,184    /* .tolerance = */ 1e-6f,185    /* .max_log_elements = */ 4096186};187 188static ggml_et_cpu_compare_config ssm_conv_cpu_compare_config = {189    /* .enabled = */ false,190    /* .use_cpu_result = */ false,191    /* .log_differences = */ true,192    /* .tolerance = */ 1e-6f,193    /* .max_log_elements = */ 4096194};195 196static ggml_et_cpu_compare_config rwkv_wkv6_cpu_compare_config = {197    /* .enabled = */ false,198    /* .use_cpu_result = */ false,199    /* .log_differences = */ true,200    /* .tolerance = */ 1e-4f,201    /* .max_log_elements = */ 4096202};203 204static ggml_et_cpu_compare_config rwkv_wkv7_cpu_compare_config = {205    /* .enabled = */ false,206    /* .use_cpu_result = */ false,207    /* .log_differences = */ true,208    /* .tolerance = */ 1e-4f,209    /* .max_log_elements = */ 4096210};211 212static ggml_et_cpu_compare_config set_rows_cpu_compare_config = {213    /* .enabled = */ false,214    /* .use_cpu_result = */ false,215    /* .log_differences = */ true,216    /* .tolerance = */ 1e-6f,217    /* .max_log_elements = */ 2048218};219 220bool ggml_et_op_rms_norm_mul(ggml_backend_et_device_context * dev_ctx,221                             const ggml_tensor *              rms_norm_node,222                             const ggml_tensor *              mul_node) {223    ET_PERF_START();224 225    if (!dev_ctx || !rms_norm_node || !mul_node) {226        GGML_LOG_ERROR("ET: Invalid parameters for fused RMS_NORM_MUL operation\n");227        return false;228    }229 230    if (!rms_norm_node->src[0]) {231        GGML_LOG_ERROR("ET: Fused RMS_NORM_MUL missing required input\n");232        return false;233    }234 235    // Extract weights: the MUL operand that isn't the rms_norm output236    const ggml_tensor * weights = (mul_node->src[0] == rms_norm_node) ? mul_node->src[1] : mul_node->src[0];237 238    if (!weights) {239        GGML_LOG_ERROR("ET: Fused RMS_NORM_MUL missing weights tensor\n");240        return false;241    }242 243    float eps;244    memcpy(&eps, rms_norm_node->op_params, sizeof(float));245 246    ggml_et_rms_norm_mul_params params;247    params.src0 = *rms_norm_node->src[0];  // input to normalize248    params.src1 = *weights;                // normalization weights249    params.dst  = *mul_node;               // final output250    params.eps  = eps;251 252    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "rms_norm_mul_f32", &params, sizeof(params), 0xFFFFFFFF);253 254    ET_PERF_END_EXT("RMS_NORM_MUL", "rms_norm_mul_f32", mul_node, "eps=%.6f", (double) eps);255    return kernel_result;256}257 258bool ggml_et_op_scale(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {259    ET_PERF_START();260 261    if (!dev_ctx || !node) {262        GGML_LOG_ERROR("ET: Invalid parameters for SCALE operation\n");263        return false;264    }265 266    if (!node->src[0]) {267        GGML_LOG_ERROR("ET: SCALE operation missing required input\n");268        return false;269    }270 271    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {272        GGML_LOG_ERROR("ET: SCALE operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),273                       ggml_type_name(node->src[0]->type));274        return false;275    }276 277    float scale, bias;278    memcpy(&scale, (const float *) node->op_params + 0, sizeof(float));279    memcpy(&bias, (const float *) node->op_params + 1, sizeof(float));280 281    ggml_et_scale_params params;282    params.src0  = *node->src[0];283    params.dst   = *node;284    params.scale = scale;285    params.bias  = bias;286 287    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "scale_f32", &params, sizeof(params), 0xFFFFFFFF);288 289    ET_PERF_END_EXT("SCALE", "scale_f32", node, "scale=%.6f|bias=%.6f", (double) scale, (double) bias);290    return kernel_result;291}292 293bool ggml_et_op_sqr(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {294    ET_PERF_START();295 296    if (!dev_ctx || !node) {297        GGML_LOG_ERROR("ET: Invalid parameters for SQR operation\n");298        return false;299    }300 301    if (!node->src[0]) {302        GGML_LOG_ERROR("ET: SQR operation missing required input\n");303        return false;304    }305 306    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {307        GGML_LOG_ERROR("ET: SQR operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),308                       ggml_type_name(node->src[0]->type));309        return false;310    }311 312    ggml_et_sqr_params params;313    params.src0 = *node->src[0];  // F32 input tensor314    params.dst  = *node;          // F32 output tensor315 316    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)317    ggml_et_cpu_compare_ctx cpu_cmp_ctx;318    bool                    cpu_comparison_active = false;319    if (sqr_cpu_compare_config.enabled) {320        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_SQR)) {321            cpu_comparison_active = true;322        } else {323            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for SQR operation\n");324        }325    }326 327    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "sqr_f32", &params, sizeof(params), 0xFFFFFFFF);328 329    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)330    if (cpu_comparison_active) {331        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &sqr_cpu_compare_config)) {332            GGML_LOG_WARN("ET: CPU comparison failed for SQR operation\n");333        }334        ggml_et_cpu_compare_free(&cpu_cmp_ctx);335    }336 337    ET_PERF_END("SQR", "sqr_f32", node);338    return kernel_result;339}340 341bool ggml_et_op_sum_rows(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {342    ET_PERF_START();343 344    if (!dev_ctx || !node) {345        GGML_LOG_ERROR("ET: Invalid parameters for SUM_ROWS operation\n");346        return false;347    }348 349    if (!node->src[0]) {350        GGML_LOG_ERROR("ET: SUM_ROWS operation missing required input\n");351        return false;352    }353 354    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {355        GGML_LOG_ERROR("ET: SUM_ROWS operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),356                       ggml_type_name(node->src[0]->type));357        return false;358    }359 360    ggml_et_sum_rows_params params;361    params.src0 = *node->src[0];362    params.dst  = *node;363 364    // Phase 1: Initialize CPU comparison context365    ggml_et_cpu_compare_ctx cpu_cmp_ctx;366    bool                    cpu_comparison_active = false;367    if (sum_rows_cpu_compare_config.enabled) {368        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_SUM_ROWS)) {369            cpu_comparison_active = true;370        } else {371            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for SUM_ROWS operation\n");372        }373    }374 375    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "sum_rows_f32", &params, sizeof(params), 0xFFFFFFFF);376 377    // Phase 2: Execute CPU computation and compare378    if (cpu_comparison_active) {379        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &sum_rows_cpu_compare_config)) {380            GGML_LOG_WARN("ET: CPU comparison failed for SUM_ROWS operation\n");381        }382        ggml_et_cpu_compare_free(&cpu_cmp_ctx);383    }384 385    ET_PERF_END("SUM_ROWS", "sum_rows_f32", node);386    return kernel_result;387}388 389bool ggml_et_op_mean(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {390    ET_PERF_START();391 392    if (!dev_ctx || !node) {393        GGML_LOG_ERROR("ET: Invalid parameters for MEAN operation\n");394        return false;395    }396 397    if (!node->src[0]) {398        GGML_LOG_ERROR("ET: MEAN operation missing required input\n");399        return false;400    }401 402    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {403        GGML_LOG_ERROR("ET: MEAN operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),404                       ggml_type_name(node->src[0]->type));405        return false;406    }407 408    ggml_et_mean_params params;409    params.src0 = *node->src[0];410    params.dst  = *node;411 412    ggml_et_cpu_compare_ctx cpu_cmp_ctx;413    bool                    cpu_comparison_active = false;414    if (mean_cpu_compare_config.enabled) {415        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_MEAN)) {416            cpu_comparison_active = true;417        } else {418            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for MEAN operation\n");419        }420    }421 422    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "mean_f32", &params, sizeof(params), 0xFFFFFFFF);423 424    if (cpu_comparison_active) {425        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &mean_cpu_compare_config)) {426            GGML_LOG_WARN("ET: CPU comparison failed for MEAN operation\n");427        }428        ggml_et_cpu_compare_free(&cpu_cmp_ctx);429    }430 431    ET_PERF_END("MEAN", "mean_f32", node);432    return kernel_result;433}434 435bool ggml_et_op_clamp(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {436    ET_PERF_START();437 438    if (!dev_ctx || !node) {439        GGML_LOG_ERROR("ET: Invalid parameters for CLAMP operation\n");440        return false;441    }442 443    if (!node->src[0]) {444        GGML_LOG_ERROR("ET: CLAMP operation missing required input\n");445        return false;446    }447 448    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {449        GGML_LOG_ERROR("ET: CLAMP operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),450                       ggml_type_name(node->src[0]->type));451        return false;452    }453 454    ggml_et_clamp_params params;455    params.src0 = *node->src[0];456    params.dst  = *node;457    // op_params layout per ggml.c::ggml_clamp: { min, max } as floats458    memcpy(&params.min_val, (const float *) node->op_params + 0, sizeof(float));459    memcpy(&params.max_val, (const float *) node->op_params + 1, sizeof(float));460 461    ggml_et_cpu_compare_ctx cpu_cmp_ctx;462    bool                    cpu_comparison_active = false;463    if (clamp_cpu_compare_config.enabled) {464        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_CLAMP)) {465            cpu_comparison_active = true;466        } else {467            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for CLAMP operation\n");468        }469    }470 471    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "clamp_f32", &params, sizeof(params), 0xFFFFFFFF);472 473    if (cpu_comparison_active) {474        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &clamp_cpu_compare_config)) {475            GGML_LOG_WARN("ET: CPU comparison failed for CLAMP operation\n");476        }477        ggml_et_cpu_compare_free(&cpu_cmp_ctx);478    }479 480    ET_PERF_END("CLAMP", "clamp_f32", node);481    return kernel_result;482}483 484bool ggml_et_op_unary(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {485    ET_PERF_START();486 487    if (!dev_ctx || !node) {488        GGML_LOG_ERROR("ET: Invalid parameters for UNARY operation\n");489        return false;490    }491 492    if (!node->src[0]) {493        GGML_LOG_ERROR("ET: UNARY operation missing required input\n");494        return false;495    }496 497    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {498        GGML_LOG_ERROR("ET: UNARY operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),499                       ggml_type_name(node->src[0]->type));500        return false;501    }502 503    const ggml_unary_op uop     = ggml_get_unary_op(node);504    const char *        op_name = ggml_unary_op_name(uop);505 506    ggml_et_unary_params params;507    params.src0     = *node->src[0];  // F32 input tensor508    params.dst      = *node;          // F32 output tensor509    params.unary_op = (int32_t) uop;510 511    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)512    ggml_et_cpu_compare_ctx cpu_cmp_ctx;513    bool                    cpu_comparison_active = false;514    if (unary_cpu_compare_config.enabled) {515        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_UNARY)) {516            cpu_comparison_active = true;517        } else {518            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for UNARY/%s operation\n", op_name);519        }520    }521 522    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "unary_f32", &params, sizeof(params), 0xFFFFFFFF);523 524    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)525    if (cpu_comparison_active) {526        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &unary_cpu_compare_config)) {527            GGML_LOG_WARN("ET: CPU comparison failed for UNARY/%s operation\n", op_name);528        }529        ggml_et_cpu_compare_free(&cpu_cmp_ctx);530    }531 532    ET_PERF_END_EXT("UNARY", "unary_f32", node, "op=%s", op_name);533    return kernel_result;534}535 536bool ggml_et_op_mul(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {537    // Delegate to generic element map operation538    return ggml_et_op_elmap(dev_ctx, node);539}540 541bool ggml_et_op_add(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {542    // Delegate to generic element map operation543    return ggml_et_op_elmap(dev_ctx, node);544}545 546bool ggml_et_op_sub(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {547    // Delegate to generic element map operation548    return ggml_et_op_elmap(dev_ctx, node);549}550 551bool ggml_et_op_elmap(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {552    ET_PERF_START();553 554    if (!dev_ctx || !node) {555        GGML_LOG_ERROR("ET: Invalid parameters for element map operation\n");556        return false;557    }558 559    if (!node->src[0] || !node->src[1]) {560        GGML_LOG_ERROR("ET: Element map operation missing required inputs\n");561        return false;562    }563 564    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32 || node->src[1]->type != GGML_TYPE_F32) {565        GGML_LOG_ERROR("ET: Element map operation with unsupported types: dst=%s src0=%s src1=%s\n",566                       ggml_type_name(node->type), ggml_type_name(node->src[0]->type),567                       ggml_type_name(node->src[1]->type));568        return false;569    }570 571    const char * op_name = ggml_op_name(node->op);572 573    ggml_et_elmap_params params;574    params.src0 = *node->src[0];575    params.src1 = *node->src[1];576    params.dst  = *node;  // F32 output tensor (op type stored in dst.op)577 578    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)579    ggml_et_cpu_compare_ctx cpu_cmp_ctx;580    bool                    cpu_comparison_active = false;581    if (elmap_cpu_compare_config.enabled) {582        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, node->op)) {583            cpu_comparison_active = true;584        } else {585            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for %s operation\n", op_name);586        }587    }588 589    // fprintf(stderr, "ET: el_map s0 [%ld, %ld, %ld, %ld] s1 [%ld, %ld, %ld, %ld]\n",590    //     node->src[0]->ne[0], node->src[0]->ne[1], node->src[0]->ne[2], node->src[0]->ne[3],591    //     node->src[1]->ne[0], node->src[1]->ne[1], node->src[1]->ne[2], node->src[1]->ne[3]);592 593    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "el_map_f32", &params, sizeof(params), 0xFFFFFFFF);594 595    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)596    if (cpu_comparison_active) {597        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &elmap_cpu_compare_config)) {598            GGML_LOG_WARN("ET: CPU comparison failed for %s operation\n", op_name);599        }600        ggml_et_cpu_compare_free(&cpu_cmp_ctx);601    }602 603    ET_PERF_END(op_name, "el_map_f32", node);604    return kernel_result;605}606 607bool ggml_et_op_glu(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {608    ET_PERF_START();609 610    // Validate inputs611    if (!dev_ctx || !node) {612        GGML_LOG_ERROR("ET: Invalid parameters for GLU operation\n");613        return false;614    }615 616    if (!node->src[0]) {617        GGML_LOG_ERROR("ET: GLU operation missing required input\n");618        return false;619    }620 621    const bool is_split_mode = node->src[1] != nullptr;622 623    // Only support F32 (as validated by supports_op)624    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32 ||625        (is_split_mode && node->src[1]->type != GGML_TYPE_F32)) {626        return false;627    }628 629    // Extract GLU operation parameters from op_params630    int32_t glu_op_type = ggml_get_op_params_i32(node, 0);  // GLU variant (REGLU, GEGLU, SWIGLU, etc.)631    int32_t swapped     = ggml_get_op_params_i32(node, 1);  // Whether gate/value are swapped632 633    // Supported variants634    switch (glu_op_type) {635        case GGML_GLU_OP_REGLU:636        case GGML_GLU_OP_GEGLU:637        case GGML_GLU_OP_SWIGLU:638        case GGML_GLU_OP_SWIGLU_OAI:639        case GGML_GLU_OP_GEGLU_ERF:640        case GGML_GLU_OP_GEGLU_QUICK:641            break;642        default:643            GGML_LOG_ERROR("ET: GLU operation with unsupported variant: %s\n",644                           ggml_glu_op_name((ggml_glu_op) glu_op_type));645            return false;646    }647 648    // Get GLU operation name for logging649    const char * glu_op_name = ggml_glu_op_name((ggml_glu_op) glu_op_type);650 651    // Pack parameters. Single-tensor mode is encoded by zeroing src1.652    ggml_et_glu_params params = {};653    params.src0               = *node->src[0];654    if (is_split_mode) {655        params.src1 = *node->src[1];656    }657    params.dst         = *node;658    params.glu_op_type = glu_op_type;659    params.swapped     = swapped;660    params.alpha       = 0.0f;661    params.limit       = 0.0f;662    if (glu_op_type == GGML_GLU_OP_SWIGLU_OAI) {663        params.alpha = ggml_get_op_params_f32(node, 2);664        params.limit = ggml_get_op_params_f32(node, 3);665    }666    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)667    ggml_et_cpu_compare_ctx cpu_cmp_ctx;668    bool                    cpu_comparison_active = false;669    if (glu_cpu_compare_config.enabled) {670        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_GLU)) {671            cpu_comparison_active = true;672        } else {673            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for %s operation\n", glu_op_name);674        }675    }676 677    // Launch ET kernel678    bool kernel_result = ggml_et_launch_kernel(dev_ctx, "glu_f32", &params, sizeof(params), 0xFFFFFFFF);679 680    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)681    if (cpu_comparison_active) {682        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &glu_cpu_compare_config)) {683            GGML_LOG_WARN("ET: CPU comparison failed for %s operation\n", glu_op_name);684        }685        ggml_et_cpu_compare_free(&cpu_cmp_ctx);686    }687 688    ET_PERF_END("GLU", "glu_f32", node);689    return kernel_result;690}691 692bool ggml_et_op_mul_mat(ggml_backend_et_device_context * dev_ctx,693                        const ggml_tensor *              node,694                        const ggml_tensor *              add_node) {695    ET_PERF_START();696 697    if (!dev_ctx || !node) {698        GGML_LOG_ERROR("ET: Invalid parameters for MUL_MAT operation\n");699        return false;700    }701 702    if (!node->src[0] || !node->src[1]) {703        GGML_LOG_ERROR("ET: MUL_MAT operation missing required inputs\n");704        return false;705    }706 707    // Fused MM+ADD: when add_node is non-NULL the caller has already validated708    // (Q8_0 weights, F32 acts, exact-shape ADD with stride parity to dst) via709    // ggml_et_can_fuse({MUL_MAT, ADD}). The kernel writes dst = mm + bias and710    // the ADD's output replaces MM's as the actual dst.711    const ggml_tensor * fused_dst   = add_node ? add_node : node;712    const ggml_tensor * bias_tensor = nullptr;713    if (add_node) {714        bias_tensor = (add_node->src[0] == node) ? add_node->src[1] : add_node->src[0];715    }716 717    const char * kernel_name;718    const char * src0_type_name;719 720    if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q4_0 && node->src[1]->type == GGML_TYPE_F32 &&721        node->src[1]->ne[1] >= 53 &&      // N >= 53722        node->src[0]->ne[1] % 16 == 0 &&  // M % TILE_M723        node->src[0]->ne[0] % 32 == 0) {  // K % BLOCK_K (Q4_0 block)724 725        // Matrix engine for N >= 53; partial N (via n_cur-1) and errata padding are handled in-kernel.726        kernel_name    = "mul_mat_Q4_0_matrix_engine";727        src0_type_name = "Q4_0";728 729    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q4_0 &&730               node->src[1]->type == GGML_TYPE_F32) {731        kernel_name    = "mul_mat_Q4_0";  // N < 53, or M % 16 != 0 or K % 32 != 0732        src0_type_name = "Q4_0";733 734    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q8_0 &&735               node->src[1]->type == GGML_TYPE_F32) {736        kernel_name    = "mul_mat_Q8_0";737        src0_type_name = "Q8_0";738 739    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F16 &&740               node->src[1]->type == GGML_TYPE_F16 && node->ne[0] % 16 == 0 && node->src[0]->ne[0] % 16 == 0 &&741               node->src[0]->ne[1] % 16 == 0 && node->src[1]->ne[0] != 1) {742        kernel_name    = "mul_mat_f16_matrix_engine";743        src0_type_name = "F16";744 745    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F16 &&746               (node->src[1]->type == GGML_TYPE_F16 || node->src[1]->type == GGML_TYPE_F32)) {747        kernel_name    = "mul_mat_f16";748        src0_type_name = "F16";749 750    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32 &&751               node->src[1]->type == GGML_TYPE_F32 && node->ne[0] % 16 == 0 && node->src[0]->ne[0] % 16 == 0 &&752               node->src[0]->ne[1] % 16 == 0 && node->src[1]->ne[0] != 1) {  // GEMV is faster with the generic path753 754        kernel_name    = "mul_mat_f32_matrix_engine";755        src0_type_name = "F32";756    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32 &&757               (node->src[1]->type == GGML_TYPE_F16 || node->src[1]->type == GGML_TYPE_F32)) {758        kernel_name    = "mul_mat_f32";759        src0_type_name = "F32";760    } else {761        GGML_LOG_ERROR("ET: MUL_MAT operation with unsupported types: dst=%s src0=%s src1=%s\n",762                       ggml_type_name(node->type), ggml_type_name(node->src[0]->type),763                       ggml_type_name(node->src[1]->type));764        return false;765    }766 767    ggml_et_binary_params params;768    params.src0 = *node->src[0];  // weight matrix769    params.src1 = *node->src[1];  // activation matrix770    params.dst  = *fused_dst;     // output (= add_node when fused, else node)771 772    ggml_et_cpu_compare_ctx cpu_cmp_ctx;773    bool                    cpu_comparison_active = false;774    if (mul_mat_cpu_compare_config.enabled) {775        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, fused_dst, GGML_OP_MUL_MAT)) {776            cpu_comparison_active = true;777        } else {778            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for MUL_MAT operation\n");779        }780    }781 782    bool kernel_result;783    if (node->src[0]->type == GGML_TYPE_Q8_0) {784        // Q8_0 kernel always takes the extended struct. bias.data is non-NULL785        // only on the fused path; otherwise the kernel skips the add entirely.786        ggml_et_mm_q8_params q8_params = {};787        q8_params.src0                 = params.src0;788        q8_params.src1                 = params.src1;789        q8_params.dst                  = params.dst;790        if (bias_tensor) {791            q8_params.bias = *bias_tensor;792        }793        kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, &q8_params, sizeof(q8_params), 0xFFFFFFFF);794    } else {795        // Non-Q8 MM kernels don't yet support fused-add; the graph fuse check796        // already rejects non-Q8 pairs, so add_node is always nullptr here.797        kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, &params, sizeof(params), 0xFFFFFFFF);798    }799 800    // printf("Tensor error:");801    // if (params.src0.data != NULL)802    // {803    //     printf("Ptr OK\n");804    //     printf("node->data ptr = %p\n", node->data);805    //     // if (once < 100){806    //     //     // uint64_t * host_data = (uint64_t *) node->data;807    //     //     // printf("Tensor error: %lu\n", host_data[0]);808 809    //     //     // printf("Tensor error:");810    //     //     once++;811    //     // }812    // }813 814    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)815    if (cpu_comparison_active) {816        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, fused_dst, &mul_mat_cpu_compare_config)) {817            GGML_LOG_WARN("ET: CPU comparison failed for MUL_MAT operation\n");818        }819        ggml_et_cpu_compare_free(&cpu_cmp_ctx);820    }821 822    {823        // Calculate actual FLOPs including batch/sequence dimensions824        // dst shape: [M, N, ne2, ne3] where M=ne[1], N=ne[0]825        int64_t m   = node->ne[1];826        int64_t n   = node->ne[0];827        int64_t k   = node->src[0]->ne[0];828        int64_t ne2 = node->ne[2];829        int64_t ne3 = node->ne[3];830 831        // Total FLOPs = (batch_size) * M * N * (2*K - 1)832        // Each MxN matrix-matrix multiply does M*N*(2*K-1) FLOPs833        // Broadcasting is handled by repeating computation, so count actual operations834        int64_t batch_size  = ne2 * ne3;835        int64_t total_flops = batch_size * m * n * (2 * k - 1);836 837        char kernel_variant[64];838        snprintf(kernel_variant, sizeof(kernel_variant), "%s_%sx%s", kernel_name, src0_type_name,839                 ggml_type_name(node->src[1]->type));840        ET_PERF_END_EXT("MUL_MAT", kernel_variant, node, "flops=%" PRId64, total_flops);841    }842    return kernel_result;843}844 845bool ggml_et_op_mul_mat_id(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {846    ET_PERF_START();847    if (!dev_ctx || !node) {848        GGML_LOG_ERROR("ET: Invalid parameters for MUL_MAT_ID operation\n");849        return false;850    }851 852    if (!node->src[0] || !node->src[1] || !node->src[2]) {853        GGML_LOG_ERROR("ET: MUL_MAT_ID operation missing required inputs\n");854        return false;855    }856 857    const char * kernel_name;858    const char * src0_type_name;859 860    // Support Q8_0/Q4_0/F16/F32 x F32 -> F32 matrix multiplication with expert selection861    if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q8_0 && node->src[1]->type == GGML_TYPE_F32 &&862        node->src[2]->type == GGML_TYPE_I32) {863        kernel_name    = "mul_mat_id_Q8_0";864        src0_type_name = "Q8_0";865 866    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_Q4_0 &&867               node->src[1]->type == GGML_TYPE_F32 && node->src[2]->type == GGML_TYPE_I32) {868        kernel_name    = "mul_mat_id_Q4_0";869        src0_type_name = "Q4_0";870 871    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F16 &&872               node->src[1]->type == GGML_TYPE_F32 && node->src[2]->type == GGML_TYPE_I32) {873        kernel_name    = "mul_mat_id_f32";874        src0_type_name = "F16";875 876    } else if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32 &&877               node->src[1]->type == GGML_TYPE_F32 && node->src[2]->type == GGML_TYPE_I32) {878        kernel_name    = "mul_mat_id_f32";879        src0_type_name = "F32";880 881    } else {882        GGML_LOG_ERROR("ET: MUL_MAT_ID operation with unsupported types: dst=%s src0=%s src1=%s src2=%s\n",883                       ggml_type_name(node->type), ggml_type_name(node->src[0]->type),884                       ggml_type_name(node->src[1]->type), ggml_type_name(node->src[2]->type));885        return false;886    }887 888    // Pack parameters - copy full tensor structures889    ggml_et_mul_mat_id_params params;890    params.src0 = *node->src[0];  // Expert weight matrices (Q8_0/F16/F32)891    params.src1 = *node->src[1];  // Activation matrix (F32)892    params.src2 = *node->src[2];  // Expert indices (I32)893    params.dst  = *node;          // Output matrix (F32)894 895    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)896    ggml_et_cpu_compare_ctx cpu_cmp_ctx;897    bool                    cpu_comparison_active = false;898    if (mul_mat_id_cpu_compare_config.enabled) {899        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_MUL_MAT_ID)) {900            cpu_comparison_active = true;901        } else {902            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for MUL_MAT_ID operation\n");903        }904    }905 906    // Launch ET kernel907    bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, &params, sizeof(params), 0xFFFFFFFF);908 909    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)910    if (cpu_comparison_active) {911        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &mul_mat_id_cpu_compare_config)) {912            GGML_LOG_WARN("ET: CPU comparison failed for MUL_MAT_ID operation\n");913        }914        ggml_et_cpu_compare_free(&cpu_cmp_ctx);915    }916 917    // Calculate FLOPs (approximate - similar to MUL_MAT but with expert routing overhead)918    // Each expert computation is similar to a MUL_MAT, but we only compute for selected experts919    int64_t K             = node->src[0]->ne[0];920    int64_t M             = node->src[0]->ne[1];921    int64_t n_expert_used = node->src[2]->ne[0];922    int64_t batch         = node->src[2]->ne[1];923 924    int64_t total_flops = batch * n_expert_used * M * (2 * K - 1);925 926    char kernel_variant[64];927    snprintf(kernel_variant, sizeof(kernel_variant), "%s_%sx%s", kernel_name, src0_type_name,928             ggml_type_name(node->src[1]->type));929    ET_PERF_END_EXT("MUL_MAT_ID", kernel_variant, node, "flops=%" PRId64 "|n_expert=%lld|n_expert_used=%lld",930                    total_flops, (long long) node->src[0]->ne[2], (long long) n_expert_used);931 932    return kernel_result;933}934 935bool ggml_et_op_rope(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {936    ET_PERF_START();937 938    if (!dev_ctx || !node) {939        GGML_LOG_ERROR("ET: Invalid parameters for ROPE operation\n");940        return false;941    }942 943    if (!node->src[0] || !node->src[1]) {944        GGML_LOG_ERROR("ET: ROPE operation missing required inputs\n");945        return false;946    }947 948    const char * kernel_name;949 950    if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32 && node->src[1]->type == GGML_TYPE_I32) {951        kernel_name = "rope_f32";952    } else {953        return false;954    }955 956    // Pack parameters - copy full tensor structures and op_params957    ggml_et_rope_params params;958    params.src0 = *node->src[0];                       // F32 input tensor959    params.src1 = *node->src[1];                       // I32 position tensor960    if (node->src[2]) {961        params.src2 = *node->src[2];                   // F32 frequency factors (optional)962    } else {963        memset(&params.src2, 0, sizeof(params.src2));  // Zero if not provided964    }965    params.dst = *node;                                // F32 output tensor966 967    params.rope_params.n_past     = ((const int32_t *) node->op_params)[0];968    params.rope_params.n_dims     = ((const int32_t *) node->op_params)[1];969    params.rope_params.mode       = ((const int32_t *) node->op_params)[2];970    params.rope_params.n_ctx      = ((const int32_t *) node->op_params)[3];971    params.rope_params.n_ctx_orig = ((const int32_t *) node->op_params)[4];972    memcpy(&params.rope_params.freq_base, (const int32_t *) node->op_params + 5, sizeof(float));973    memcpy(&params.rope_params.freq_scale, (const int32_t *) node->op_params + 6, sizeof(float));974    memcpy(&params.rope_params.ext_factor, (const int32_t *) node->op_params + 7, sizeof(float));975    memcpy(&params.rope_params.attn_factor, (const int32_t *) node->op_params + 8, sizeof(float));976    memcpy(&params.rope_params.beta_fast, (const int32_t *) node->op_params + 9, sizeof(float));977    memcpy(&params.rope_params.beta_slow, (const int32_t *) node->op_params + 10, sizeof(float));978    if (params.rope_params.mode & GGML_ROPE_TYPE_MROPE) {979        memcpy(params.rope_params.sections, (const int32_t *) node->op_params + 11, sizeof(int32_t) * 4);980    } else {981        memset(params.rope_params.sections, 0, sizeof(params.rope_params.sections));982    }983 984    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)985    ggml_et_cpu_compare_ctx cpu_cmp_ctx;986    bool                    cpu_comparison_active = false;987    if (rope_cpu_compare_config.enabled) {988        GGML_LOG_DEBUG("ET: Initializing CPU comparison for ROPE operation\n");989        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_ROPE)) {990            cpu_comparison_active = true;991        } else {992            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for ROPE operation\n");993        }994    }995 996    bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, &params, sizeof(params), 0xFFFFFFFF);997 998    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)999    if (cpu_comparison_active) {1000        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &rope_cpu_compare_config)) {1001            GGML_LOG_WARN("ET: CPU comparison failed for ROPE operation\n");1002        }1003        ggml_et_cpu_compare_free(&cpu_cmp_ctx);1004    }1005 1006    ET_PERF_END_EXT("ROPE", kernel_name, node, "mode=0x%x|n_dims=%d|freq_base=%.2f|freq_scale=%.2f",1007                    params.rope_params.mode, params.rope_params.n_dims, (double) params.rope_params.freq_base,1008                    (double) params.rope_params.freq_scale);1009    return kernel_result;1010}1011 1012bool ggml_et_op_rms_norm(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {1013    ET_PERF_START();1014 1015    if (!dev_ctx || !node) {1016        GGML_LOG_ERROR("ET: Invalid parameters for RMS_NORM operation\n");1017        return false;1018    }1019 1020    if (!node->src[0]) {1021        GGML_LOG_ERROR("ET: RMS_NORM operation missing required input\n");1022        return false;1023    }1024 1025    const char * kernel_name;1026 1027    if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32) {1028        kernel_name = "rms_norm_f32";1029 1030    } else {1031        GGML_LOG_ERROR("ET: RMS_NORM operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),1032                       ggml_type_name(node->src[0]->type));1033        return false;1034    }1035 1036    float eps;1037    memcpy(&eps, node->op_params, sizeof(float));1038 1039    ggml_et_rms_norm_params params;1040    params.src0 = *node->src[0];  // F32 input tensor1041    params.dst  = *node;          // F32 output tensor1042    params.eps  = eps;            // Epsilon parameter for numerical stability1043 1044    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)1045    ggml_et_cpu_compare_ctx cpu_cmp_ctx;1046    bool                    cpu_comparison_active = false;1047    if (rms_norm_cpu_compare_config.enabled) {1048        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_RMS_NORM)) {1049            cpu_comparison_active = true;1050        } else {1051            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for RMS_NORM operation\n");1052        }1053    }1054 1055    bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, &params, sizeof(params), 0xFFFFFFFF);1056 1057    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)1058    if (cpu_comparison_active) {1059        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &rms_norm_cpu_compare_config)) {1060            GGML_LOG_WARN("ET: CPU comparison failed for RMS_NORM operation\n");1061        }1062        ggml_et_cpu_compare_free(&cpu_cmp_ctx);1063    }1064 1065    ET_PERF_END_EXT("RMS_NORM", kernel_name, node, "eps=%.6f", (double) eps);1066    return kernel_result;1067}1068 1069bool ggml_et_op_norm(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {1070    ET_PERF_START();1071 1072    if (!dev_ctx || !node) {1073        GGML_LOG_ERROR("ET: Invalid parameters for NORM operation\n");1074        return false;1075    }1076 1077    if (!node->src[0]) {1078        GGML_LOG_ERROR("ET: NORM operation missing required input\n");1079        return false;1080    }1081 1082    const char * kernel_name;1083 1084    if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32) {1085        kernel_name = "norm_f32";1086 1087    } else {1088        GGML_LOG_ERROR("ET: NORM operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),1089                       ggml_type_name(node->src[0]->type));1090        return false;1091    }1092 1093    float eps;1094    memcpy(&eps, node->op_params, sizeof(float));1095 1096    ggml_et_norm_params params;1097    params.src0 = *node->src[0];  // F32 input tensor1098    params.dst  = *node;          // F32 output tensor1099    params.eps  = eps;            // Epsilon parameter for numerical stability1100 1101    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)1102    ggml_et_cpu_compare_ctx cpu_cmp_ctx;1103    bool                    cpu_comparison_active = false;1104    if (norm_cpu_compare_config.enabled) {1105        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_NORM)) {1106            cpu_comparison_active = true;1107        } else {1108            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for NORM operation\n");1109        }1110    }1111 1112    bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, &params, sizeof(params), 0xFFFFFFFF);1113 1114    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)1115    if (cpu_comparison_active) {1116        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &norm_cpu_compare_config)) {1117            GGML_LOG_WARN("ET: CPU comparison failed for NORM operation\n");1118        }1119        ggml_et_cpu_compare_free(&cpu_cmp_ctx);1120    }1121 1122    ET_PERF_END_EXT("NORM", kernel_name, node, "eps=%.6f", (double) eps);1123    return kernel_result;1124}1125 1126bool ggml_et_op_l2_norm(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {1127    ET_PERF_START();1128 1129    if (!dev_ctx || !node) {1130        GGML_LOG_ERROR("ET: Invalid parameters for L2_NORM operation\n");1131        return false;1132    }1133 1134    if (!node->src[0]) {1135        GGML_LOG_ERROR("ET: L2_NORM operation missing required input\n");1136        return false;1137    }1138 1139    const char * kernel_name;1140 1141    if (node->type == GGML_TYPE_F32 && node->src[0]->type == GGML_TYPE_F32) {1142        kernel_name = "l2_norm_f32";1143 1144    } else {1145        GGML_LOG_ERROR("ET: L2_NORM operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),1146                       ggml_type_name(node->src[0]->type));1147        return false;1148    }1149 1150    float eps;1151    memcpy(&eps, node->op_params, sizeof(float));1152 1153    ggml_et_l2_norm_params params;1154    params.src0 = *node->src[0];  // F32 input tensor1155    params.dst  = *node;          // F32 output tensor1156    params.eps  = eps;            // Epsilon parameter for numerical stability1157 1158    // Phase 1: Initialize CPU comparison context and copy source buffers (before ET kernel)1159    ggml_et_cpu_compare_ctx cpu_cmp_ctx;1160    bool                    cpu_comparison_active = false;1161    if (l2_norm_cpu_compare_config.enabled) {1162        if (ggml_et_cpu_compare_init_pre(&cpu_cmp_ctx, node, GGML_OP_L2_NORM)) {1163            cpu_comparison_active = true;1164        } else {1165            GGML_LOG_WARN("ET: Failed to initialize CPU comparison for L2_NORM operation\n");1166        }1167    }1168 1169    bool kernel_result = ggml_et_launch_kernel(dev_ctx, kernel_name, &params, sizeof(params), 0xFFFFFFFF);1170 1171    // Phase 2: Execute CPU computation and compare with ET result (after ET kernel)1172    if (cpu_comparison_active) {1173        if (!ggml_et_cpu_compare_compute_and_check(&cpu_cmp_ctx, node, &l2_norm_cpu_compare_config)) {1174            GGML_LOG_WARN("ET: CPU comparison failed for L2_NORM operation\n");1175        }1176        ggml_et_cpu_compare_free(&cpu_cmp_ctx);1177    }1178 1179    ET_PERF_END_EXT("L2_NORM", kernel_name, node, "eps=%.6f", (double) eps);1180    return kernel_result;1181}1182 1183bool ggml_et_op_group_norm(ggml_backend_et_device_context * dev_ctx, const ggml_tensor * node) {1184    ET_PERF_START();1185 1186    if (!dev_ctx || !node) {1187        GGML_LOG_ERROR("ET: Invalid parameters for GROUP_NORM operation\n");1188        return false;1189    }1190 1191    if (!node->src[0]) {1192        GGML_LOG_ERROR("ET: GROUP_NORM operation missing required input\n");1193        return false;1194    }1195 1196    if (node->type != GGML_TYPE_F32 || node->src[0]->type != GGML_TYPE_F32) {1197        GGML_LOG_ERROR("ET: GROUP_NORM operation with unsupported types: dst=%s src0=%s\n", ggml_type_name(node->type),1198                       ggml_type_name(node->src[0]->type));1199        return false;1200    }

Showing the first 1,200 of 2581 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai