Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
ggml-et-cpu-compare.cpp498 linesDownload Raw Back to ggml-et
1#include "ggml-et-cpu-compare.h"2 3#include "ggml-cpu/ggml-cpu-impl.h"4#include "ggml-cpu/ops.h"5 6#include <algorithm>7#include <cmath>8#include <cstdlib>9#include <cstring>10 11bool ggml_et_cpu_compare_init_pre(ggml_et_cpu_compare_ctx * ctx, const ggml_tensor * node, ggml_op op) {12    if (!ctx || !node) {13        GGML_LOG_ERROR("ET: Invalid parameters for CPU compare init\n");14        return false;15    }16 17    // Clear context18    memset(ctx, 0, sizeof(*ctx));19 20    // Calculate actual buffer sizes - use backend buffer size for accurate copy21    auto get_tensor_buffer_size = [](const ggml_tensor * tensor) -> size_t {22        if (!tensor) {23            return 0;24        }25 26        if (tensor->buffer) {27            // Get actual backend buffer size28            size_t buffer_size = ggml_backend_buffer_get_size(tensor->buffer);29 30            // Use the full buffer size to avoid any truncation issues31            return buffer_size;32        } else {33            // Fallback to logical size if no buffer34            return ggml_nbytes(tensor);35        }36    };37 38    ctx->src0_size = get_tensor_buffer_size(node->src[0]);39    ctx->src1_size = get_tensor_buffer_size(node->src[1]);40    ctx->src2_size = get_tensor_buffer_size(node->src[2]);41    ctx->dst_size  = get_tensor_buffer_size(node);42 43    // Allocate CPU buffers for all tensors44    if (ctx->src0_size > 0) {45        ctx->cpu_src0_data = malloc(ctx->src0_size);46        if (!ctx->cpu_src0_data) {47            GGML_LOG_ERROR("ET: Failed to allocate CPU src0 buffer\n");48            goto cleanup;49        }50    }51 52    if (ctx->src1_size > 0) {53        ctx->cpu_src1_data = malloc(ctx->src1_size);54        if (!ctx->cpu_src1_data) {55            GGML_LOG_ERROR("ET: Failed to allocate CPU src1 buffer\n");56            goto cleanup;57        }58    }59 60    if (ctx->src2_size > 0) {61        ctx->cpu_src2_data = malloc(ctx->src2_size);62        if (!ctx->cpu_src2_data) {63            GGML_LOG_ERROR("ET: Failed to allocate CPU src2 buffer\n");64            goto cleanup;65        }66    }67 68    ctx->cpu_dst_data = malloc(ctx->dst_size);69    if (!ctx->cpu_dst_data) {70        GGML_LOG_ERROR("ET: Failed to allocate CPU dst buffer\n");71        goto cleanup;72    }73 74    ctx->et_dst_data = malloc(ctx->dst_size);75    if (!ctx->et_dst_data) {76        GGML_LOG_ERROR("ET: Failed to allocate ET dst buffer\n");77        goto cleanup;78    }79 80    // Copy data from ET device buffers to CPU host buffers81    if (ctx->src0_size > 0) {82        // Copy logical tensor size - ggml_backend_tensor_get handles stride layout internally83        size_t logical_size = ggml_nbytes(node->src[0]);84        ggml_backend_tensor_get(node->src[0], ctx->cpu_src0_data, 0, logical_size);85    }86    if (ctx->src1_size > 0) {87        size_t logical_size = ggml_nbytes(node->src[1]);88        ggml_backend_tensor_get(node->src[1], ctx->cpu_src1_data, 0, logical_size);89    }90    if (ctx->src2_size > 0) {91        size_t logical_size = ggml_nbytes(node->src[2]);92        ggml_backend_tensor_get(node->src[2], ctx->cpu_src2_data, 0, logical_size);93    }94 95    // Copy destination data from device (for operations like SET_ROWS that modify existing data)96    // Most ops create new tensors so this is unused, but SET_ROWS requires existing dst data97    {98        size_t logical_size = ggml_nbytes(node);99        ggml_backend_tensor_get(node, ctx->cpu_dst_data, 0, logical_size);100    }101 102    // Create CPU backend for reference computation103    GGML_LOG_DEBUG("ET: Creating CPU backend for reference computation\n");104    ctx->cpu_backend = ggml_backend_cpu_init();105    if (!ctx->cpu_backend) {106        GGML_LOG_ERROR("ET: Failed to create CPU backend\n");107        goto cleanup;108    }109 110    // Create GGML context for CPU tensors111    GGML_LOG_DEBUG("ET: Creating GGML context for CPU computation\n");112    ggml_init_params ctx_params;113    ctx_params.mem_size   = ggml_tensor_overhead() * 4 + ggml_graph_overhead();  // up to 4 tensors + graph114    ctx_params.mem_buffer = nullptr;115    ctx_params.no_alloc   = true;                                                // We'll manage data ourselves116    ctx->ggml_ctx         = ggml_init(ctx_params);117    if (!ctx->ggml_ctx) {118        GGML_LOG_ERROR("ET: Failed to create GGML context\n");119        goto cleanup;120    }121 122    // Create CPU tensors with proper context123    if (node->src[0]) {124        ctx->cpu_src0 = ggml_new_tensor(ctx->ggml_ctx, node->src[0]->type, GGML_MAX_DIMS, node->src[0]->ne);125        if (!ctx->cpu_src0) {126            GGML_LOG_ERROR("ET: Failed to create CPU src0 tensor\n");127            goto cleanup;128        }129        ctx->cpu_src0->data = ctx->cpu_src0_data;130        // Copy stride array (nb) for correct memory layout131        memcpy(ctx->cpu_src0->nb, node->src[0]->nb, sizeof(node->src[0]->nb));132        // Copy op_params if present133        memcpy(ctx->cpu_src0->op_params, node->src[0]->op_params, sizeof(node->src[0]->op_params));134    }135 136    if (node->src[1]) {137        ctx->cpu_src1 = ggml_new_tensor(ctx->ggml_ctx, node->src[1]->type, GGML_MAX_DIMS, node->src[1]->ne);138        if (!ctx->cpu_src1) {139            GGML_LOG_ERROR("ET: Failed to create CPU src1 tensor\n");140            goto cleanup;141        }142        ctx->cpu_src1->data = ctx->cpu_src1_data;143        // Copy stride array (nb) for correct memory layout144        memcpy(ctx->cpu_src1->nb, node->src[1]->nb, sizeof(node->src[1]->nb));145        // Copy op_params if present146        memcpy(ctx->cpu_src1->op_params, node->src[1]->op_params, sizeof(node->src[1]->op_params));147    }148 149    if (node->src[2]) {150        ctx->cpu_src2 = ggml_new_tensor(ctx->ggml_ctx, node->src[2]->type, GGML_MAX_DIMS, node->src[2]->ne);151        if (!ctx->cpu_src2) {152            GGML_LOG_ERROR("ET: Failed to create CPU src2 tensor\n");153            goto cleanup;154        }155        ctx->cpu_src2->data = ctx->cpu_src2_data;156        // Copy stride array (nb) for correct memory layout157        memcpy(ctx->cpu_src2->nb, node->src[2]->nb, sizeof(node->src[2]->nb));158        // Copy op_params if present159        memcpy(ctx->cpu_src2->op_params, node->src[2]->op_params, sizeof(node->src[2]->op_params));160    }161 162    return true;163 164cleanup:165    ggml_et_cpu_compare_free(ctx);166    return false;167}168 169bool ggml_et_cpu_compare_compute_and_check(ggml_et_cpu_compare_ctx *          ctx,170                                           const ggml_tensor *                node,171                                           const ggml_et_cpu_compare_config * config) {172    if (!ctx || !ctx->cpu_backend || !ctx->ggml_ctx || !node || !config) {173        GGML_LOG_ERROR("ET: Invalid parameters for CPU compute and check\n");174        return false;175    }176 177    // Create operation-specific CPU destination tensor based on the node's operation178    ggml_op op = node->op;179    switch (op) {180        case GGML_OP_MUL:181            ctx->cpu_dst = ggml_mul(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1);182            break;183        case GGML_OP_ADD:184            ctx->cpu_dst = ggml_add(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1);185            break;186        case GGML_OP_MUL_MAT:187            ctx->cpu_dst = ggml_mul_mat(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1);188            break;189        case GGML_OP_MUL_MAT_ID:190            // MUL_MAT_ID: Mixture of Experts matrix multiplication191            // src0 (as): expert weight matrices [K, M, n_expert]192            // src1 (b):  activations [K, n_expert_used, batch]193            // src2 (ids): expert selection indices [n_expert_used, batch]194            ctx->cpu_dst = ggml_mul_mat_id(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, ctx->cpu_src2);195            break;196        case GGML_OP_ROPE:197            {198                const int32_t * op_params  = (const int32_t *) node->op_params;199                const int32_t   n_dims     = op_params[1];200                const int32_t   mode       = op_params[2];201                const int32_t   n_ctx_orig = op_params[4];202                const float     freq_base  = *((const float *) (op_params + 5));203                const float     freq_scale = *((const float *) (op_params + 6));204                const float     ext_factor = *((const float *) (op_params + 7));205                const float     attn_factor = *((const float *) (op_params + 8));206                const float     beta_fast  = *((const float *) (op_params + 9));207                const float     beta_slow  = *((const float *) (op_params + 10));208 209                if (mode & GGML_ROPE_TYPE_MROPE) {210                    int sections[GGML_MROPE_SECTIONS];211                    memcpy(sections, op_params + 11, sizeof(sections));212                    ctx->cpu_dst = ggml_rope_multi(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, ctx->cpu_src2,213                                                   n_dims, sections, mode, n_ctx_orig, freq_base, freq_scale,214                                                   ext_factor, attn_factor, beta_fast, beta_slow);215                } else {216                    ctx->cpu_dst = ggml_rope_ext(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, ctx->cpu_src2,217                                                 n_dims, mode, n_ctx_orig, freq_base, freq_scale, ext_factor,218                                                 attn_factor, beta_fast, beta_slow);219                }220            }221            break;222        case GGML_OP_RMS_NORM:223            // Extract epsilon parameter from op_params (stored as float)224            {225                float eps;226                memcpy(&eps, node->op_params, sizeof(float));227                ctx->cpu_dst = ggml_rms_norm(ctx->ggml_ctx, ctx->cpu_src0, eps);228            }229            break;230        case GGML_OP_SQR:231            ctx->cpu_dst = ggml_sqr(ctx->ggml_ctx, ctx->cpu_src0);232            break;233        case GGML_OP_UNARY:234            {235                ggml_unary_op uop = (ggml_unary_op) ggml_get_op_params_i32(node, 0);236                ctx->cpu_dst      = ggml_unary(ctx->ggml_ctx, ctx->cpu_src0, uop);237            }238            break;239        case GGML_OP_SUM_ROWS:240            ctx->cpu_dst = ggml_sum_rows(ctx->ggml_ctx, ctx->cpu_src0);241            break;242        case GGML_OP_MEAN:243            ctx->cpu_dst = ggml_mean(ctx->ggml_ctx, ctx->cpu_src0);244            break;245        case GGML_OP_CLAMP:246            {247                float clamp_min, clamp_max;248                memcpy(&clamp_min, (const float *) node->op_params + 0, sizeof(float));249                memcpy(&clamp_max, (const float *) node->op_params + 1, sizeof(float));250                ctx->cpu_dst = ggml_clamp(ctx->ggml_ctx, ctx->cpu_src0, clamp_min, clamp_max);251            }252            break;253        case GGML_OP_GLU:254            // Extract GLU parameters from op_params (split mode only)255            {256                int32_t     glu_op_type = ggml_get_op_params_i32(node, 0);  // GLU variant257                ggml_glu_op glu_op      = (ggml_glu_op) glu_op_type;258 259                // Only support split tensor mode260                if (!ctx->cpu_src1) {261                    GGML_LOG_ERROR("ET: GLU CPU comparison requires split tensor mode\n");262                    return false;263                }264                ctx->cpu_dst = ggml_glu_split(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, glu_op);265            }266            break;267        case GGML_OP_SOFT_MAX:268            {269                // Extract scale and max_bias from op_params270                float scale    = 1.0f;271                float max_bias = 0.0f;272                memcpy(&scale, (const float *) node->op_params + 0, sizeof(float));273                memcpy(&max_bias, (const float *) node->op_params + 1, sizeof(float));274 275                if (ctx->cpu_src1 || scale != 1.0f || max_bias != 0.0f) {276                    // Use extended softmax when mask or non-default parameters are present277                    ctx->cpu_dst = ggml_soft_max_ext(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1, scale, max_bias);278                } else {279                    // Use simple softmax when no mask and default parameters280                    ctx->cpu_dst = ggml_soft_max(ctx->ggml_ctx, ctx->cpu_src0);281                }282 283                // Add sinks if present284                if (ctx->cpu_src2) {285                    ggml_soft_max_add_sinks(ctx->cpu_dst, ctx->cpu_src2);286                }287            }288            break;289        case GGML_OP_GET_ROWS:290            ctx->cpu_dst = ggml_get_rows(ctx->ggml_ctx, ctx->cpu_src0, ctx->cpu_src1);291            break;292        case GGML_OP_CONT:293            ctx->cpu_dst = ggml_cont(ctx->ggml_ctx, ctx->cpu_src0);294            break;295        case GGML_OP_SET_ROWS:296            {297                // SET_ROWS operation scatters src0 rows to dst[src1] positions298                // Create destination tensor (this is the "view" that SET_ROWS returns)299                ggml_tensor * cpu_dst_base = ggml_new_tensor(ctx->ggml_ctx, node->type, GGML_MAX_DIMS, node->ne);300                if (!cpu_dst_base) {301                    GGML_LOG_ERROR("ET: Failed to create CPU destination base tensor for SET_ROWS\n");302                    return false;303                }304                cpu_dst_base->data = ctx->cpu_dst_data;305                memcpy(cpu_dst_base->nb, node->nb, sizeof(node->nb));306 307                // Note: cpu_dst_data already contains the pre-existing destination data from device308                // SET_ROWS will update specific rows, leaving others unchanged309 310                // Perform SET_ROWS operation: returns a view that scatters src0 rows to dst[src1] positions311                ctx->cpu_dst = ggml_set_rows(ctx->ggml_ctx, cpu_dst_base, ctx->cpu_src0, ctx->cpu_src1);312            }313            break;314        default:315            GGML_LOG_ERROR("ET: Unsupported operation %s for CPU comparison\n", ggml_op_name(op));316            return false;317    }318 319    if (!ctx->cpu_dst) {320        GGML_LOG_ERROR("ET: Failed to create CPU destination tensor for operation %s\n", ggml_op_name(op));321        return false;322    }323 324    ctx->cpu_dst->data = ctx->cpu_dst_data;325    // Copy stride array (nb) for correct memory layout - except for CONT which should keep contiguous strides326    if (op != GGML_OP_CONT) {327        memcpy(ctx->cpu_dst->nb, node->nb, sizeof(node->nb));328    }329    // For CONT operations, keep the contiguous strides created by ggml_cont()330 331    // Create minimal computation graph332    ctx->cpu_graph = ggml_new_graph_custom(ctx->ggml_ctx, 1, false);333    if (!ctx->cpu_graph) {334        GGML_LOG_ERROR("ET: Failed to create CPU computation graph\n");335        return false;336    }337    ctx->cpu_graph->nodes[0] = ctx->cpu_dst;338    ctx->cpu_graph->n_nodes  = 1;339 340    // Log input data for debugging if enabled341    if (config && config->log_differences) {342        if (ctx->cpu_src0_data && ctx->src0_size >= 4) {343            GGML_LOG_DEBUG("ET: CPU src0 first few bytes: %02x %02x %02x %02x\n", ((uint8_t *) ctx->cpu_src0_data)[0],344                           ((uint8_t *) ctx->cpu_src0_data)[1], ((uint8_t *) ctx->cpu_src0_data)[2],345                           ((uint8_t *) ctx->cpu_src0_data)[3]);346        }347        if (ctx->cpu_src1_data && ctx->src1_size >= 16) {348            GGML_LOG_DEBUG("ET: CPU src1 first few floats: %.6f %.6f %.6f %.6f\n", ((float *) ctx->cpu_src1_data)[0],349                           ((float *) ctx->cpu_src1_data)[1], ((float *) ctx->cpu_src1_data)[2],350                           ((float *) ctx->cpu_src1_data)[3]);351        }352    }353 354    // Compute using CPU backend355    ggml_status cpu_result = ggml_backend_graph_compute(ctx->cpu_backend, ctx->cpu_graph);356 357    if (cpu_result != GGML_STATUS_SUCCESS) {358        GGML_LOG_ERROR("ET: CPU reference computation failed with status %d\n", cpu_result);359        return false;360    }361 362    // Log output data for debugging if enabled363    if (config && config->log_differences && ctx->dst_size >= 16) {364        GGML_LOG_DEBUG("ET: CPU dst first few floats after computation: %.6f %.6f %.6f %.6f\n",365                       ((float *) ctx->cpu_dst_data)[0], ((float *) ctx->cpu_dst_data)[1],366                       ((float *) ctx->cpu_dst_data)[2], ((float *) ctx->cpu_dst_data)[3]);367    }368 369    // Now copy ET device destination to host for comparison370    size_t dst_logical_size = ggml_nbytes(node);371    ggml_backend_tensor_get(node, ctx->et_dst_data, 0, dst_logical_size);372 373    if (config->log_differences) {374        size_t num_elements = ggml_nelements(node);375        size_t max_log      = std::min(num_elements, config->max_log_elements);376 377        // Check if this is an elementwise operation that can show src inputs378        bool    is_elementwise = (op == GGML_OP_MUL || op == GGML_OP_ADD || op == GGML_OP_GLU);379        float * cpu_src0_float = is_elementwise ? (float *) ctx->cpu_src0_data : nullptr;380        float * cpu_src1_float = is_elementwise ? (float *) ctx->cpu_src1_data : nullptr;381 382        // Helper to get float value from tensor data (handles f16 and f32)383        auto get_float = [](const void * data, size_t idx, ggml_type type) -> float {384            if (type == GGML_TYPE_F16) {385                const ggml_fp16_t * fp16_data = (const ggml_fp16_t *) data;386                return ggml_fp16_to_fp32(fp16_data[idx]);387            }388 389            const float * float_data = (const float *) data;390            return float_data[idx];391        };392 393        // Compare all elements but log only the first max_log_elements394        bool   matches          = true;395        size_t total_mismatches = 0;396 397        // First pass: check all elements for mismatches398        for (size_t i = 0; i < num_elements; i++) {399            float cpu_val  = get_float(ctx->cpu_dst_data, i, node->type);400            float et_val   = get_float(ctx->et_dst_data, i, node->type);401            float diff     = fabsf(cpu_val - et_val);402            float rel_diff = diff / (fabsf(cpu_val) + 1e-8f);403 404            if (rel_diff > config->tolerance) {405                matches = false;406                total_mismatches++;407            }408        }409 410        // Second pass: log detailed info for first max_log elements only411        for (size_t i = 0; i < max_log; i++) {412            float cpu_val = get_float(ctx->cpu_dst_data, i, node->type);413            float et_val  = get_float(ctx->et_dst_data, i, node->type);414            float diff    = fabsf(cpu_val - et_val);415 416            if (is_elementwise && cpu_src0_float && cpu_src1_float) {417                GGML_LOG_DEBUG("ET: [%zu] src0=%.6f, src1=%.6f -> CPU=%.6f, ET=%.6f, diff=%.6f\n", i, cpu_src0_float[i],418                               cpu_src1_float[i], cpu_val, et_val, diff);419            } else if (is_elementwise && cpu_src0_float) {420                GGML_LOG_DEBUG("ET: [%zu] src0=%.6f -> CPU=%.6f, ET=%.6f, diff=%.6f\n", i, cpu_src0_float[i], cpu_val,421                               et_val, diff);422            } else {423                GGML_LOG_DEBUG("ET: [%zu] CPU=%.6f, ET=%.6f, diff=%.6f\n", i, cpu_val, et_val, diff);424            }425        }426 427        // Check some elements from the middle and end for full coverage428        if (num_elements > max_log) {429            size_t mid     = num_elements / 2;430            size_t end     = num_elements - 1;431            float  cpu_mid = get_float(ctx->cpu_dst_data, mid, node->type);432            float  et_mid  = get_float(ctx->et_dst_data, mid, node->type);433            float  cpu_end = get_float(ctx->cpu_dst_data, end, node->type);434            float  et_end  = get_float(ctx->et_dst_data, end, node->type);435 436            GGML_LOG_DEBUG("ET: Middle element [%zu]: CPU=%.6f, ET=%.6f\n", mid, cpu_mid, et_mid);437            GGML_LOG_DEBUG("ET: Last element [%zu]: CPU=%.6f, ET=%.6f\n", end, cpu_end, et_end);438        }439 440        GGML_LOG_DEBUG("ET: Results %s (%zu/%zu elements match within tolerance %.6f)\n", matches ? "MATCH" : "DIFFER",441                       num_elements - total_mismatches, num_elements, config->tolerance);442    }443 444    // Copy CPU result to device if flag is set445    if (config->use_cpu_result) {446        GGML_LOG_DEBUG("ET: Overwriting ET device result with CPU result for correct inference\n");447        size_t dst_logical_size = ggml_nbytes(node);448        ggml_backend_tensor_set(const_cast<ggml_tensor *>(node), ctx->cpu_dst_data, 0, dst_logical_size);449        GGML_LOG_DEBUG("ET: CPU result copied to ET device buffer\n");450    }451 452    return true;453}454 455void ggml_et_cpu_compare_free(ggml_et_cpu_compare_ctx * ctx) {456    if (!ctx) {457        return;458    }459 460    if (ctx->cpu_src0_data) {461        free(ctx->cpu_src0_data);462        ctx->cpu_src0_data = nullptr;463    }464    if (ctx->cpu_src1_data) {465        free(ctx->cpu_src1_data);466        ctx->cpu_src1_data = nullptr;467    }468    if (ctx->cpu_src2_data) {469        free(ctx->cpu_src2_data);470        ctx->cpu_src2_data = nullptr;471    }472    if (ctx->cpu_dst_data) {473        free(ctx->cpu_dst_data);474        ctx->cpu_dst_data = nullptr;475    }476    if (ctx->et_dst_data) {477        free(ctx->et_dst_data);478        ctx->et_dst_data = nullptr;479    }480 481    if (ctx->ggml_ctx) {482        ggml_free(ctx->ggml_ctx);483        ctx->ggml_ctx = nullptr;484    }485 486    if (ctx->cpu_backend) {487        ggml_backend_free(ctx->cpu_backend);488        ctx->cpu_backend = nullptr;489    }490 491    // Clear pointers492    ctx->cpu_src0  = nullptr;493    ctx->cpu_src1  = nullptr;494    ctx->cpu_src2  = nullptr;495    ctx->cpu_dst   = nullptr;496    ctx->cpu_graph = nullptr;497}498 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai