Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
ggml-webgpu.cpp4763 linesDownload Raw Back to ggml-webgpu
1/*2    WebGPU backend implementation.3    Note: Use ClangFormat to format this file.4*/5 6#include "ggml-webgpu.h"7 8#include "ggml-backend-impl.h"9#include "ggml-impl.h"10#include "ggml-webgpu-shader-lib.hpp"11#include "ggml.h"12 13#ifdef __EMSCRIPTEN__14#    include <emscripten/emscripten.h>15#endif16 17#include <webgpu/webgpu_cpp.h>18 19#include <atomic>20#include <cstdint>21#include <cstring>22#ifdef GGML_WEBGPU_GPU_PROFILE23#    include <iomanip>24#endif25#if defined(GGML_WEBGPU_DEBUG) || defined(GGML_WEBGPU_CPU_PROFILE) || defined(GGML_WEBGPU_GPU_PROFILE)26#    include <iostream>27#endif28#include <memory>29#include <mutex>30#include <optional>31#include <string>32#include <utility>33#include <vector>34 35#define ROUNDUP_POW2(x, pow2) (((x) + ((pow2) - 1)) & ~((pow2) - 1))36#define CEIL_DIV(M, N)        (((M) + (N) - 1) / (N))37 38// Return a rectangular grid of workgroups with minimal over-provisioned workgroups.39// Assumes that the total number of workgroups does not exceed max_per_dim^2.40static inline void compute_2d_workgroups(uint32_t total_wg, uint32_t max_per_dim, uint32_t & wg_x, uint32_t & wg_y) {41    wg_y = std::max(1u, CEIL_DIV(total_wg, max_per_dim));42    wg_x = CEIL_DIV(total_wg, wg_y);43}44 45static inline uint32_t ggml_webgpu_u32_from_f32(float value) {46    uint32_t bits;47    memcpy(&bits, &value, sizeof(bits));48    return bits;49}50 51#ifdef GGML_WEBGPU_DEBUG52#    define WEBGPU_LOG_DEBUG(msg)  std::cout << msg << std::endl53#    define WEBGPU_DEBUG_BUF_ELEMS 51254#else55#    define WEBGPU_LOG_DEBUG(msg) ((void) 0)56#endif  // GGML_WEBGPU_DEBUG57 58#ifdef GGML_WEBGPU_CPU_PROFILE59// total timing (aggregated)60#    define WEBGPU_CPU_PROFILE_TOTAL_START(id) auto cpu_total_start_##id = std::chrono::high_resolution_clock::now();61 62#    define WEBGPU_CPU_PROFILE_TOTAL_END(id, ctx)                                                         \63        auto   cpu_total_end_##id = std::chrono::high_resolution_clock::now();                            \64        double cpu_total_time_##id =                                                                      \65            std::chrono::duration<double, std::milli>(cpu_total_end_##id - cpu_total_start_##id).count(); \66        (ctx)->cpu_time_ms[#id] += cpu_total_time_##id;67// fine-grained timing (not included in totals)68#    define WEBGPU_CPU_PROFILE_DETAIL_START(id) auto cpu_detail_start_##id = std::chrono::high_resolution_clock::now();69 70#    define WEBGPU_CPU_PROFILE_DETAIL_END(id, ctx)                                                          \71        auto   cpu_detail_end_##id = std::chrono::high_resolution_clock::now();                             \72        double cpu_detail_time_##id =                                                                       \73            std::chrono::duration<double, std::milli>(cpu_detail_end_##id - cpu_detail_start_##id).count(); \74        (ctx)->cpu_detail_ms[#id] += cpu_detail_time_##id;75#else76#    define WEBGPU_CPU_PROFILE_TOTAL_START(id)77#    define WEBGPU_CPU_PROFILE_TOTAL_END(id, ctx)78#    define WEBGPU_CPU_PROFILE_DETAIL_START(id)79#    define WEBGPU_CPU_PROFILE_DETAIL_END(id, ctx)80#endif  // GGML_WEBGPU_CPU_PROFILE81 82#ifdef GGML_WEBGPU_GPU_PROFILE83#    define WEBGPU_MAX_PROFILE_QUERY_COUNT        4096u84#    define WEBGPU_TIMESTAMP_QUERY_BUF_SIZE_BYTES (WEBGPU_MAX_PROFILE_QUERY_COUNT * sizeof(uint64_t))85#endif86 87/* Constants */88 89#define WEBGPU_DEFAULT_COMMAND_SUBMIT_BATCH_SIZE 64u90#define WEBGPU_NUM_PARAM_SLOT_SAFETY_MARGIN      10u91#define WEBGPU_RUNTIME_WAIT_TIMEOUT_MS           30000u92#define WEBGPU_RUNTIME_WAIT_TIMEOUT_NS           (WEBGPU_RUNTIME_WAIT_TIMEOUT_MS * 1e6)93#define WEBGPU_PARAMS_BUF_SIZE_BYTES             128  // enough for 32 parameters94#define WEBGPU_SET_ROWS_ERROR_BUF_SIZE_BYTES     495#define WEBGPU_STORAGE_BUF_BINDING_MULT          4    // a storage buffer binding size must be a multiple of 496 97/* End Constants */98 99// This is a "fake" base pointer, since WebGPU buffers do not have pointers to100// their locations.101static void * const webgpu_ptr_base = (void *) (uintptr_t) 0x1000;  // NOLINT102 103static size_t ggml_webgpu_tensor_offset(const ggml_tensor * tensor) {104    const ggml_tensor * base_tensor = tensor->view_src ? tensor->view_src : tensor;105    return (size_t) ((uintptr_t) base_tensor->data - (uintptr_t) webgpu_ptr_base) + tensor->view_offs;106}107 108/* Struct definitions */109 110// Forward reference111static void ggml_webgpu_create_buffer(wgpu::Device &    device,112                                      wgpu::Buffer &    buffer,113                                      size_t            size,114                                      wgpu::BufferUsage usage,115                                      const char *      label);116 117// Slot-based parameter arena for compute graph encoding. Each encoded kernel118// gets a unique uniform-buffer slice within the current batch, and the slot119// cursor is reset immediately after that batch is submitted.120struct webgpu_param_arena {121    wgpu::Buffer buffer;122    size_t       slot_stride = 0;123    size_t       slot_size   = 0;124    uint32_t     slot_count  = 0;125    uint32_t     next_slot   = 0;126 127    void init(wgpu::Device device, size_t slot_size, uint32_t slot_count, size_t alignment) {128        this->slot_stride = ROUNDUP_POW2(slot_size, alignment);129        this->slot_size   = slot_size;130        this->slot_count  = slot_count;131        this->next_slot   = 0;132 133        ggml_webgpu_create_buffer(device, buffer, this->slot_stride * slot_count,134                                  wgpu::BufferUsage::CopyDst | wgpu::BufferUsage::Uniform, "ggml_webgpu_param_arena");135    }136 137    size_t alloc_slot(size_t size) {138        GGML_ASSERT(size <= slot_size);139        if (next_slot >= slot_count) {140            GGML_ABORT("ggml_webgpu: parameter arena exhausted while encoding a batch");141        }142 143        return slot_stride * next_slot++;144    }145 146    void reset() { next_slot = 0; }147 148    void cleanup() {149        if (buffer) {150            buffer.Destroy();151            buffer = nullptr;152        }153    }154 155    ~webgpu_param_arena() { this->cleanup(); }156};157 158struct webgpu_encoded_op {159    uint32_t num_kernels = 0;160#ifdef GGML_WEBGPU_GPU_PROFILE161    std::vector<std::string> pipeline_names;162#endif163};164 165struct webgpu_dispatch_desc {166    webgpu_pipeline                   pipeline;167    std::vector<uint32_t>             params;168    std::vector<wgpu::BindGroupEntry> bind_group_entries;169    std::pair<uint32_t, uint32_t>     workgroups = { 1, 1 };170};171 172struct webgpu_capabilities {173    wgpu::Limits limits;174    bool         supports_subgroups       = false;175    bool         supports_subgroup_matrix = false;176    bool         supports_dot_product     = false;177 178    uint32_t sg_mat_m = 0;179    uint32_t sg_mat_n = 0;180    uint32_t sg_mat_k = 0;181 182    uint32_t subgroup_size     = 0;183    uint32_t min_subgroup_size = 0;184    uint32_t max_subgroup_size = 0;185    size_t   memset_bytes_per_thread;186};187 188// Stores global webgpu members189struct webgpu_global_context_struct {190    wgpu::Instance instance;191    wgpu::Adapter  adapter;192    wgpu::Device   device;193    wgpu::Queue    queue;194    uint32_t       command_submit_batch_size = WEBGPU_DEFAULT_COMMAND_SUBMIT_BATCH_SIZE;195    uint32_t       max_inflight_batches      = UINT32_MAX;196 197    webgpu_capabilities  capabilities;198    // Shared buffer to move data from device to host199    wgpu::Buffer         get_tensor_staging_buf;200    // Global mutex for get_tensor201    std::recursive_mutex mutex;202 203    wgpu::Buffer    memset_params_buf;204    webgpu_pipeline memset_pipeline;205 206    std::string vendor;207 208    // TODO: We should rework the CPU profiling time handling to make it more useful. ref: https://github.com/ggml-org/llama.cpp/pull/22050209#ifdef GGML_WEBGPU_CPU_PROFILE210    // Profiling: labeled CPU time in ms (total)211    std::unordered_map<std::string, double> cpu_time_ms;212    // Profiling: detailed CPU time in ms213    std::unordered_map<std::string, double> cpu_detail_ms;214#endif215 216#ifdef GGML_WEBGPU_DEBUG217    wgpu::Buffer debug_host_buf;218    wgpu::Buffer debug_dev_buf;219#endif220 221    ~webgpu_global_context_struct() {222        if (this->get_tensor_staging_buf) {223            this->get_tensor_staging_buf.Destroy();224            this->get_tensor_staging_buf = nullptr;225        }226        if (this->memset_params_buf) {227            this->memset_params_buf.Destroy();228            this->memset_params_buf = nullptr;229        }230#ifdef GGML_WEBGPU_DEBUG231        if (this->debug_host_buf) {232            this->debug_host_buf.Destroy();233            this->debug_host_buf = nullptr;234        }235        if (this->debug_dev_buf) {236            this->debug_dev_buf.Destroy();237            this->debug_dev_buf = nullptr;238        }239#endif240    }241};242 243typedef std::shared_ptr<webgpu_global_context_struct> webgpu_global_context;244 245// All the base objects needed to run operations on a WebGPU device246struct webgpu_context_struct {247    // Points to global instances owned by ggml_backend_webgpu_reg_context248    webgpu_global_context global_ctx;249 250    std::unique_ptr<ggml_webgpu_shader_lib> shader_lib;251 252    webgpu_param_arena       param_arena;253    wgpu::Buffer             set_rows_dev_error_buf;254    wgpu::Buffer             set_rows_host_error_buf;255    wgpu::CommandEncoder     active_command_encoder;256    wgpu::ComputePassEncoder active_compute_pass;257    bool                     batch_compute_passes = true;258 259    size_t memset_bytes_per_thread;260 261#ifdef GGML_WEBGPU_GPU_PROFILE262    // Profiling: per-shader GPU time in ms263    std::unordered_map<std::string, double> shader_gpu_time_ms;264    wgpu::Buffer                            profile_timestamp_dev_buf;265    wgpu::Buffer                            profile_timestamp_host_buf;266    wgpu::QuerySet                          profile_timestamp_query_set;267    uint32_t                                profile_timestamp_query_count = 0;268#endif269 270    ~webgpu_context_struct() {271#ifdef GGML_WEBGPU_GPU_PROFILE272        if (this->profile_timestamp_host_buf) {273            this->profile_timestamp_host_buf.Destroy();274            this->profile_timestamp_host_buf = nullptr;275        }276        if (this->profile_timestamp_dev_buf) {277            this->profile_timestamp_dev_buf.Destroy();278            this->profile_timestamp_dev_buf = nullptr;279        }280        if (this->profile_timestamp_query_set) {281            this->profile_timestamp_query_set.Destroy();282            this->profile_timestamp_query_set = nullptr;283        }284#endif285        if (this->set_rows_host_error_buf) {286            this->set_rows_host_error_buf.Destroy();287            this->set_rows_host_error_buf = nullptr;288        }289        if (this->set_rows_dev_error_buf) {290            this->set_rows_dev_error_buf.Destroy();291            this->set_rows_dev_error_buf = nullptr;292        }293    }294};295 296typedef std::shared_ptr<webgpu_context_struct> webgpu_context;297 298// Metadata required for the ggml backend registration/discovery interface299struct ggml_backend_webgpu_reg_context {300    // Since the Instance is a global entrypoint into the WebGPU API, it lives here301    webgpu_global_context webgpu_global_ctx;302    size_t                device_count;303    const char *          name;304};305 306// Per-device struct for the global logical device interface307struct ggml_backend_webgpu_device_context {308    webgpu_global_context webgpu_global_ctx;309    std::string           device_name;310    std::string           device_desc;311};312 313// Per-thread data required to actually run WebGPU operations in a backend instance314struct ggml_backend_webgpu_context {315    webgpu_context webgpu_ctx;316    std::string    name;317};318 319// Per-thread data related to buffers320struct ggml_backend_webgpu_buffer_context {321    wgpu::Buffer          buffer;322    std::string           label;323    webgpu_global_context global_ctx;324 325    ggml_backend_webgpu_buffer_context(wgpu::Buffer buf, std::string lbl, webgpu_global_context global_ctx_) :326        buffer(std::move(buf)),327        label(std::move(lbl)),328        global_ctx(std::move(global_ctx_)) {}329};330 331/* WebGPU object initializations */332 333static webgpu_pipeline ggml_webgpu_create_pipeline(wgpu::Device &                           device,334                                                   const char *                             shader_code,335                                                   const char *                             label,336                                                   const std::vector<wgpu::ConstantEntry> & constants = {}) {337    wgpu::ShaderSourceWGSL shader_source;338    shader_source.code = shader_code;339 340    wgpu::ShaderModuleDescriptor shader_desc;341    shader_desc.nextInChain = &shader_source;342 343    wgpu::ShaderModule shader_module = device.CreateShaderModule(&shader_desc);344 345    wgpu::ComputePipelineDescriptor pipeline_desc;346    pipeline_desc.label              = label;347    pipeline_desc.compute.module     = shader_module;348    pipeline_desc.compute.entryPoint = "main";   // Entry point in the WGSL code349    pipeline_desc.layout             = nullptr;  // nullptr means auto layout350    if (constants.size() > 0) {351        pipeline_desc.compute.constants     = constants.data();352        pipeline_desc.compute.constantCount = constants.size();353    }354    return { device.CreateComputePipeline(&pipeline_desc), label };355}356 357static void ggml_webgpu_create_buffer(wgpu::Device &    device,358                                      wgpu::Buffer &    buffer,359                                      size_t            size,360                                      wgpu::BufferUsage usage,361                                      const char *      label) {362    wgpu::BufferDescriptor buffer_desc;363    buffer_desc.size             = size;364    buffer_desc.usage            = usage;365    buffer_desc.label            = label;366    buffer_desc.mappedAtCreation = false;367 368    // TODO: error handling369    buffer = device.CreateBuffer(&buffer_desc);370}371 372static wgpu::Buffer ggml_webgpu_tensor_buf(const ggml_tensor * tensor) {373    ggml_backend_webgpu_buffer_context * ctx = (ggml_backend_webgpu_buffer_context *) tensor->buffer->context;374    return ctx->buffer;375}376 377static size_t ggml_webgpu_tensor_misalignment(const ggml_tensor * t, size_t alignment) {378    size_t offset = ggml_webgpu_tensor_offset(t);379    return offset & (alignment - 1);380}381 382static size_t ggml_webgpu_tensor_misalignment(webgpu_context & ctx, const ggml_tensor * t) {383    return ggml_webgpu_tensor_misalignment(t, ctx->global_ctx->capabilities.limits.minStorageBufferOffsetAlignment);384}385 386static size_t ggml_webgpu_tensor_align_offset(const ggml_tensor * t, size_t alignment) {387    size_t offset = ggml_webgpu_tensor_offset(t);388    return offset & ~(alignment - 1);389}390 391static size_t ggml_webgpu_tensor_align_offset(webgpu_context & ctx, const ggml_tensor * t) {392    return ggml_webgpu_tensor_align_offset(t, ctx->global_ctx->capabilities.limits.minStorageBufferOffsetAlignment);393}394 395static size_t ggml_webgpu_tensor_binding_size(const ggml_tensor * t, size_t alignment) {396    return ROUNDUP_POW2(ggml_nbytes(t) + ggml_webgpu_tensor_misalignment(t, alignment),397                        WEBGPU_STORAGE_BUF_BINDING_MULT);398}399 400static size_t ggml_webgpu_tensor_binding_size(webgpu_context & ctx, const ggml_tensor * t) {401    return ggml_webgpu_tensor_binding_size(t, ctx->global_ctx->capabilities.limits.minStorageBufferOffsetAlignment);402}403 404static bool ggml_webgpu_tensor_binding_overlap(const webgpu_global_context & global_ctx,405                                               const ggml_tensor *           a,406                                               const ggml_tensor *           b) {407    if (a->buffer != b->buffer) {408        return false;409    }410 411    const size_t alignment = global_ctx->capabilities.limits.minStorageBufferOffsetAlignment;412    const size_t a_offset  = ggml_webgpu_tensor_align_offset(a, alignment);413    const size_t b_offset  = ggml_webgpu_tensor_align_offset(b, alignment);414    return a_offset < b_offset + ggml_webgpu_tensor_binding_size(b, alignment) &&415           b_offset < a_offset + ggml_webgpu_tensor_binding_size(a, alignment);416}417 418static bool ggml_webgpu_tensor_binding_overlap_range(const webgpu_global_context & global_ctx,419                                                     ggml_tensor *                 tensor,420                                                     ggml_backend_buffer_t         buffer,421                                                     size_t                        offset,422                                                     size_t                        size) {423    if (tensor->buffer != buffer) {424        return false;425    }426 427    const size_t alignment     = global_ctx->capabilities.limits.minStorageBufferOffsetAlignment;428    const size_t tensor_offset = ggml_webgpu_tensor_align_offset(tensor, alignment);429    return tensor_offset < offset + size && offset < tensor_offset + ggml_webgpu_tensor_binding_size(tensor, alignment);430}431 432struct ggml_webgpu_merged_binding_range {433    size_t offset;434    size_t size;435};436 437static ggml_webgpu_merged_binding_range ggml_webgpu_tensor_merged_binding_range(438    webgpu_context &                     ctx,439    std::initializer_list<ggml_tensor *> tensors) {440    size_t merged_offset = SIZE_MAX;441    size_t merged_end    = 0;442 443    for (ggml_tensor * tensor : tensors) {444        const size_t bind_offset = ggml_webgpu_tensor_align_offset(ctx, tensor);445        const size_t bind_end    = bind_offset + ggml_webgpu_tensor_binding_size(ctx, tensor);446 447        merged_offset = std::min(merged_offset, bind_offset);448        merged_end    = std::max(merged_end, bind_end);449    }450 451    return { merged_offset, merged_end - merged_offset };452}453 454static uint32_t ggml_webgpu_tensor_merged_element_offset(const ggml_tensor *                      tensor,455                                                         const ggml_webgpu_merged_binding_range & merged_range) {456    return (uint32_t) ((ggml_webgpu_tensor_offset(tensor) - merged_range.offset) / ggml_type_size(tensor->type));457}458 459static wgpu::BindGroupEntry ggml_webgpu_make_bind_group_entry(uint32_t     binding,460                                                              wgpu::Buffer buffer,461                                                              uint64_t     offset,462                                                              uint64_t     size) {463    wgpu::BindGroupEntry entry = {};464    entry.binding              = binding;465    entry.buffer               = std::move(buffer);466    entry.offset               = offset;467    entry.size                 = size;468    return entry;469}470 471static wgpu::BindGroupEntry ggml_webgpu_make_tensor_bind_group_entry(webgpu_context & ctx,472                                                                     uint32_t         binding,473                                                                     ggml_tensor *    tensor) {474    return ggml_webgpu_make_bind_group_entry(binding, ggml_webgpu_tensor_buf(tensor),475                                             ggml_webgpu_tensor_align_offset(ctx, tensor),476                                             ggml_webgpu_tensor_binding_size(ctx, tensor));477}478 479/** End WebGPU object initializations */480 481/** WebGPU Actions */482 483template <typename T>484static void ggml_backend_webgpu_check_wait_status(wgpu::WaitStatus wait_status,485                                                  T                callback_status,486                                                  T                success_status,487                                                  const char *     wait_name,488                                                  const char *     failure_name,489                                                  const char *     callback_message) {490    if (wait_status == wgpu::WaitStatus::TimedOut) {491        GGML_ABORT("ggml_webgpu: %s timed out after %u ms\n", wait_name, WEBGPU_RUNTIME_WAIT_TIMEOUT_MS);492    }493    if (wait_status == wgpu::WaitStatus::Error) {494        GGML_ABORT("ggml_webgpu: %s failed\n", wait_name);495    }496    if (callback_status != success_status) {497        GGML_ABORT("ggml_webgpu: %s failed with status %d: %s\n", failure_name, static_cast<int>(callback_status),498                   callback_message);499    }500}501 502// TODO: these next two functions may want tuning across different platforms and workloads,503static uint32_t ggml_backend_webgpu_get_max_inflight_batches() {504    return UINT32_MAX;505}506 507static uint32_t ggml_backend_webgpu_get_command_submit_batch_size() {508    return WEBGPU_DEFAULT_COMMAND_SUBMIT_BATCH_SIZE;509}510 511static void ggml_backend_webgpu_wait_queue(webgpu_global_context & ctx) {512    wgpu::QueueWorkDoneStatus callback_status = wgpu::QueueWorkDoneStatus::Error;513    std::string               callback_message;514 515    const wgpu::WaitStatus wait_status = ctx->instance.WaitAny(516        ctx->queue.OnSubmittedWorkDone(517            wgpu::CallbackMode::AllowSpontaneous,518            [&callback_status, &callback_message](wgpu::QueueWorkDoneStatus status, wgpu::StringView message) {519                callback_status  = status;520                callback_message = std::string(message);521            }),522        WEBGPU_RUNTIME_WAIT_TIMEOUT_NS);523 524    ggml_backend_webgpu_check_wait_status(wait_status, callback_status, wgpu::QueueWorkDoneStatus::Success,525                                          "Queue wait", "Queue work", callback_message.c_str());526}527 528static void ggml_backend_webgpu_map_buffer(webgpu_global_context & ctx,529                                           wgpu::Buffer &          buffer,530                                           wgpu::MapMode           mode,531                                           size_t                  offset,532                                           size_t                  size) {533    wgpu::MapAsyncStatus callback_status = wgpu::MapAsyncStatus::Error;534    std::string          callback_message;535 536    const wgpu::WaitStatus wait_status = ctx->instance.WaitAny(537        buffer.MapAsync(mode, offset, size, wgpu::CallbackMode::AllowSpontaneous,538                        [&callback_status, &callback_message](wgpu::MapAsyncStatus status, wgpu::StringView message) {539                            callback_status  = status;540                            callback_message = std::string(message);541                        }),542        WEBGPU_RUNTIME_WAIT_TIMEOUT_NS);543 544    ggml_backend_webgpu_check_wait_status(wait_status, callback_status, wgpu::MapAsyncStatus::Success,545                                          "Buffer map wait", "Buffer map", callback_message.c_str());546}547 548static void ggml_backend_webgpu_submit_commands(webgpu_context &          ctx,549                                                const wgpu::CommandBuffer commands,550                                                uint32_t &                num_inflight_batches) {551    if (num_inflight_batches >= ctx->global_ctx->max_inflight_batches) {552        ggml_backend_webgpu_wait_queue(ctx->global_ctx);553        num_inflight_batches = 0;554    }555 556    ctx->global_ctx->queue.Submit(1, &commands);557    num_inflight_batches++;558}559 560#ifdef GGML_WEBGPU_DEBUG561// This function adds debugging information to shaders, as WebGPU does not support printing directly.562// To use, add a bind group entry to the setup for the shader you are debugging, add the buffer and563// debug statements in the shader, and then call this function after encoding the commands and submitting them.564static void ggml_backend_webgpu_debug(webgpu_global_context & ctx) {565    wgpu::CommandEncoder encoder = ctx->device.CreateCommandEncoder();566    encoder.CopyBufferToBuffer(ctx->debug_dev_buf, 0, ctx->debug_host_buf, 0, ctx->debug_host_buf.GetSize());567    wgpu::CommandBuffer commands = encoder.Finish();568    ctx->queue.Submit(1, &commands);569    ggml_backend_webgpu_map_buffer(ctx, ctx->debug_host_buf, wgpu::MapMode::Read, 0, ctx->debug_host_buf.GetSize());570    const float * debug_data = (const float *) ctx->debug_host_buf.GetConstMappedRange();571    std::cout << "debug[0]: " << debug_data[0] << "\n";572    ctx->debug_host_buf.Unmap();573}574#endif575 576static webgpu_encoded_op ggml_backend_webgpu_build_multi(webgpu_context &                          ctx,577                                                         const std::vector<webgpu_dispatch_desc> & dispatches) {578    webgpu_encoded_op            result = {};579    std::vector<wgpu::BindGroup> bind_groups;580    std::vector<size_t>          param_offsets;581    result.num_kernels = dispatches.size();582 583    for (size_t i = 0; i < dispatches.size(); i++) {584        const webgpu_dispatch_desc & dispatch     = dispatches[i];585        const size_t                 param_size   = dispatch.params.size() * sizeof(uint32_t);586        const size_t                 param_offset = ctx->param_arena.alloc_slot(param_size);587 588        std::vector<wgpu::BindGroupEntry> entries            = dispatch.bind_group_entries;589        uint32_t                          params_binding_num = entries.size();590        entries.push_back(ggml_webgpu_make_bind_group_entry(params_binding_num, ctx->param_arena.buffer, param_offset,591                                                            ctx->param_arena.slot_size));592 593        wgpu::BindGroupDescriptor bind_group_desc;594        bind_group_desc.layout     = dispatch.pipeline.pipeline.GetBindGroupLayout(0);595        bind_group_desc.entryCount = entries.size();596        bind_group_desc.entries    = entries.data();597        bind_group_desc.label      = dispatch.pipeline.name.c_str();598        bind_groups.push_back(ctx->global_ctx->device.CreateBindGroup(&bind_group_desc));599        param_offsets.push_back(param_offset);600    }601 602    for (size_t i = 0; i < param_offsets.size(); i++) {603        ctx->global_ctx->queue.WriteBuffer(ctx->param_arena.buffer, param_offsets[i], dispatches[i].params.data(),604                                           dispatches[i].params.size() * sizeof(uint32_t));605    }606 607#ifdef GGML_WEBGPU_GPU_PROFILE608    for (size_t i = 0; i < dispatches.size(); i++) {609        GGML_ASSERT(ctx->profile_timestamp_query_count + 2 <= WEBGPU_MAX_PROFILE_QUERY_COUNT);610        const uint32_t query_begin = ctx->profile_timestamp_query_count++;611        const uint32_t query_end   = ctx->profile_timestamp_query_count++;612 613        wgpu::PassTimestampWrites ts_writes   = {};614        ts_writes.querySet                    = ctx->profile_timestamp_query_set;615        ts_writes.beginningOfPassWriteIndex   = query_begin;616        ts_writes.endOfPassWriteIndex         = query_end;617        wgpu::ComputePassDescriptor pass_desc = {};618        pass_desc.timestampWrites             = &ts_writes;619 620        wgpu::ComputePassEncoder pass = ctx->active_command_encoder.BeginComputePass(&pass_desc);621 622        pass.SetPipeline(dispatches[i].pipeline.pipeline);623        pass.SetBindGroup(0, bind_groups[i]);624        pass.DispatchWorkgroups(dispatches[i].workgroups.first, dispatches[i].workgroups.second, 1);625        pass.End();626        result.pipeline_names.push_back(dispatches[i].pipeline.name);627    }628#else629    for (size_t i = 0; i < dispatches.size(); i++) {630        if (ctx->batch_compute_passes) {631            ctx->active_compute_pass.SetPipeline(dispatches[i].pipeline.pipeline);632            ctx->active_compute_pass.SetBindGroup(0, bind_groups[i]);633            ctx->active_compute_pass.DispatchWorkgroups(dispatches[i].workgroups.first, dispatches[i].workgroups.second,634                                                        1);635        } else {636            wgpu::ComputePassEncoder pass = ctx->active_command_encoder.BeginComputePass();637            pass.SetPipeline(dispatches[i].pipeline.pipeline);638            pass.SetBindGroup(0, bind_groups[i]);639            pass.DispatchWorkgroups(dispatches[i].workgroups.first, dispatches[i].workgroups.second, 1);640            pass.End();641        }642    }643#endif644 645    return result;646}647 648static webgpu_encoded_op ggml_backend_webgpu_build(webgpu_context &                  ctx,649                                                   webgpu_pipeline &                 pipeline,650                                                   std::vector<uint32_t>             params,651                                                   std::vector<wgpu::BindGroupEntry> bind_group_entries,652                                                   uint32_t                          wg_x,653                                                   uint32_t                          wg_y = 1) {654    return ggml_backend_webgpu_build_multi(655        ctx, {656                 { pipeline, std::move(params), std::move(bind_group_entries), { wg_x, wg_y } },657    });658}659 660static void ggml_backend_webgpu_buffer_memset(webgpu_global_context & ctx,661                                              wgpu::Buffer &          buf,662                                              uint32_t                value,663                                              size_t                  offset,664                                              size_t                  size) {665    std::vector<uint32_t>             params  = { (uint32_t) offset, (uint32_t) size, value };666    std::vector<wgpu::BindGroupEntry> entries = { ggml_webgpu_make_bind_group_entry(0, buf, 0, buf.GetSize()) };667    size_t                            bytes_per_wg =668        ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup * ctx->capabilities.memset_bytes_per_thread;669    uint32_t wg_x = CEIL_DIV(size + 3, bytes_per_wg);670 671    ctx->queue.WriteBuffer(ctx->memset_params_buf, 0, params.data(), params.size() * sizeof(uint32_t));672 673    wgpu::BindGroupEntry params_entry = {};674    params_entry.binding              = 1;675    params_entry.buffer               = ctx->memset_params_buf;676    params_entry.offset               = 0;677    params_entry.size                 = WEBGPU_PARAMS_BUF_SIZE_BYTES;678    entries.push_back(params_entry);679 680    wgpu::BindGroupDescriptor bind_group_desc;681    bind_group_desc.layout     = ctx->memset_pipeline.pipeline.GetBindGroupLayout(0);682    bind_group_desc.entryCount = entries.size();683    bind_group_desc.entries    = entries.data();684    bind_group_desc.label      = ctx->memset_pipeline.name.c_str();685    wgpu::BindGroup bind_group = ctx->device.CreateBindGroup(&bind_group_desc);686 687    wgpu::CommandEncoder     encoder = ctx->device.CreateCommandEncoder();688    wgpu::ComputePassEncoder pass    = encoder.BeginComputePass();689    pass.SetPipeline(ctx->memset_pipeline.pipeline);690    pass.SetBindGroup(0, bind_group);691    pass.DispatchWorkgroups(wg_x, 1, 1);692    pass.End();693 694    wgpu::CommandBuffer              command  = encoder.Finish();695    std::vector<wgpu::CommandBuffer> commands = { command };696    ctx->queue.Submit(commands.size(), commands.data());697}698 699/** End WebGPU Actions */700 701/** GGML Backend Interface */702 703static const char * ggml_backend_webgpu_name(ggml_backend_t backend) {704    ggml_backend_webgpu_context * ctx = (ggml_backend_webgpu_context *) backend->context;705    return ctx->name.c_str();706}707 708static void ggml_backend_webgpu_free(ggml_backend_t backend) {709    ggml_backend_webgpu_context * ctx = (ggml_backend_webgpu_context *) backend->context;710    WEBGPU_LOG_DEBUG("ggml_backend_webgpu_free(" << ctx->name << ")");711 712#ifdef GGML_WEBGPU_CPU_PROFILE713    std::cout << "\n[ggml_webgpu cpu profiling summary]\n";714    double total_cpu = 0.0;715    for (const auto & kv : ctx->webgpu_ctx->global_ctx->cpu_time_ms) {716        total_cpu += kv.second;717    }718    std::cout << "ggml_webgpu: total cpu time: " << total_cpu << " ms\n";719    std::cout << "ggml_webgpu: cpu breakdown:\n";720    for (const auto & kv : ctx->webgpu_ctx->global_ctx->cpu_time_ms) {721        double pct = (total_cpu > 0.0) ? (kv.second / total_cpu * 100.0) : 0.0;722        std::cout << "ggml_webgpu:  " << kv.first << ": " << kv.second << " ms (" << pct << "%)\n";723    }724    if (ctx->webgpu_ctx->global_ctx->cpu_detail_ms.size() > 0) {725        std::cout << "ggml_webgpu: cpu detailed breakdown:\n";726    }727    for (const auto & kv : ctx->webgpu_ctx->global_ctx->cpu_detail_ms) {728        double pct = (total_cpu > 0.0) ? (kv.second / total_cpu * 100.0) : 0.0;729        std::cout << "ggml_webgpu:  " << kv.first << ": " << kv.second << " ms (" << pct << "%)\n";730    }731#endif732 733#ifdef GGML_WEBGPU_GPU_PROFILE734    std::cout << "\n[ggml_webgpu gpu profiling summary]\n";735    double total_gpu = 0.0;736    for (const auto & kv : ctx->webgpu_ctx->shader_gpu_time_ms) {737        total_gpu += kv.second;738    }739    std::cout << "ggml_webgpu: total gpu time (all shaders): " << total_gpu << " ms\n";740    std::cout << "\nggml_webgpu: gpu breakdown:\n";741    for (const auto & kv : ctx->webgpu_ctx->shader_gpu_time_ms) {742        double pct = (total_gpu > 0.0) ? (kv.second / total_gpu * 100.0) : 0.0;743        std::cout << "ggml_webgpu:  " << kv.first << ": " << kv.second << " ms (" << std::fixed << std::setprecision(2)744                  << pct << "%)\n";745    }746#endif747 748#if defined(GGML_WEBGPU_CPU_PROFILE) && defined(GGML_WEBGPU_GPU_PROFILE)749    std::cout << "ggml_webgpu: gpu/cpu ratio: " << (total_cpu > 0.0 ? total_gpu / total_cpu : 0.0) << "\n";750#endif751 752    delete ctx;753    delete backend;754}755 756static webgpu_encoded_op ggml_webgpu_cpy(webgpu_context & ctx, ggml_tensor * src, ggml_tensor * dst) {757    ggml_webgpu_shader_lib_context shader_lib_ctx = {};758    shader_lib_ctx.src0                           = src;759    shader_lib_ctx.dst                            = dst;760    shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;761 762    webgpu_pipeline pipeline = ctx->shader_lib->get_cpy_pipeline(shader_lib_ctx);763 764    auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());765 766    uint32_t ne = (uint32_t) ggml_nelements(dst);767 768    std::vector<uint32_t> params = {769        ne, (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src) / ggml_type_size(src->type)),770        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),771        // Convert byte-strides to element-strides772        (uint32_t) (src->nb[0] / ggml_type_size(src->type)), (uint32_t) (src->nb[1] / ggml_type_size(src->type)),773        (uint32_t) (src->nb[2] / ggml_type_size(src->type)), (uint32_t) (src->nb[3] / ggml_type_size(src->type)),774        (uint32_t) (dst->nb[0] / ggml_type_size(dst->type)), (uint32_t) (dst->nb[1] / ggml_type_size(dst->type)),775        (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)), (uint32_t) (dst->nb[3] / ggml_type_size(dst->type)),776        // Logical shapes777        (uint32_t) src->ne[0], (uint32_t) src->ne[1], (uint32_t) src->ne[2], (uint32_t) dst->ne[0],778        (uint32_t) dst->ne[1], (uint32_t) dst->ne[2]779    };780 781    std::vector<wgpu::BindGroupEntry> entries = {782        ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src),783        ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, dst),784    };785 786    uint32_t wg_x;787    uint32_t wg_y;788    uint32_t total_wg = CEIL_DIV(ne, decisions->wg_size);789    compute_2d_workgroups(total_wg, ctx->global_ctx->capabilities.limits.maxComputeWorkgroupsPerDimension, wg_x, wg_y);790    return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);791}792 793static webgpu_encoded_op ggml_webgpu_set(webgpu_context & ctx,794                                         ggml_tensor *    src0,795                                         ggml_tensor *    src1,796                                         ggml_tensor *    dst) {797    ggml_webgpu_shader_lib_context shader_lib_ctx = {};798    shader_lib_ctx.src0                           = src0;799    shader_lib_ctx.src1                           = src1;800    shader_lib_ctx.dst                            = dst;801    shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;802 803    webgpu_pipeline pipeline = ctx->shader_lib->get_set_pipeline(shader_lib_ctx);804 805    auto *     decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());806    const bool inplace   = decisions->inplace;807 808    const uint32_t ne            = inplace ? (uint32_t) ggml_nelements(src1) : (uint32_t) ggml_nelements(dst);809    const uint32_t dst_type_size = (uint32_t) ggml_type_size(dst->type);810 811    std::vector<uint32_t> params = {812        ne,813        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),814        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),815        (uint32_t) (((const int32_t *) dst->op_params)[3] / dst_type_size),816 817        (uint32_t) (src1->nb[0] / ggml_type_size(src1->type)),818        (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)),819        (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)),820        (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)),821 822        1u,823        (uint32_t) (((const int32_t *) dst->op_params)[0] / dst_type_size),824        (uint32_t) (((const int32_t *) dst->op_params)[1] / dst_type_size),825        (uint32_t) (((const int32_t *) dst->op_params)[2] / dst_type_size),826 827        (uint32_t) src1->ne[0],828        (uint32_t) src1->ne[1],829        (uint32_t) src1->ne[2],830        (uint32_t) src1->ne[3],831    };832 833    std::vector<wgpu::BindGroupEntry> entries;834    uint32_t                          binding_index = 0;835    if (!inplace) {836        entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0));837        binding_index++;838    }839    entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, binding_index, src1));840    entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, binding_index + 1, dst));841 842    uint32_t wg_x = CEIL_DIV(ne, decisions->wg_size);843    return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x);844}845 846static webgpu_encoded_op ggml_webgpu_pad(webgpu_context & ctx, ggml_tensor * src, ggml_tensor * dst) {847    ggml_webgpu_shader_lib_context shader_lib_ctx = {};848    shader_lib_ctx.src0                           = src;849    shader_lib_ctx.dst                            = dst;850    shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;851 852    webgpu_pipeline pipeline = ctx->shader_lib->get_pad_pipeline(shader_lib_ctx);853 854    auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());855 856    const uint32_t ne = (uint32_t) ggml_nelements(dst);857 858    std::vector<uint32_t> params = {859        ne,860        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src) / ggml_type_size(src->type)),861        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),862        // Strides (in elements)863        (uint32_t) (src->nb[0] / ggml_type_size(src->type)),864        (uint32_t) (src->nb[1] / ggml_type_size(src->type)),865        (uint32_t) (src->nb[2] / ggml_type_size(src->type)),866        (uint32_t) (src->nb[3] / ggml_type_size(src->type)),867        // Shapes868        (uint32_t) src->ne[0],869        (uint32_t) src->ne[1],870        (uint32_t) src->ne[2],871        (uint32_t) src->ne[3],872        (uint32_t) dst->ne[0],873        (uint32_t) dst->ne[1],874        (uint32_t) dst->ne[2],875        (uint32_t) dst->ne[3],876        // Pad sizes877        (uint32_t) ggml_get_op_params_i32(dst, 0),878        (uint32_t) ggml_get_op_params_i32(dst, 1),879        (uint32_t) ggml_get_op_params_i32(dst, 2),880        (uint32_t) ggml_get_op_params_i32(dst, 3),881        (uint32_t) ggml_get_op_params_i32(dst, 4),882        (uint32_t) ggml_get_op_params_i32(dst, 5),883        (uint32_t) ggml_get_op_params_i32(dst, 6),884        (uint32_t) ggml_get_op_params_i32(dst, 7),885    };886 887    std::vector<wgpu::BindGroupEntry> entries = {888        ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src),889        ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, dst),890    };891 892    uint32_t wg_x = CEIL_DIV(ne, decisions->wg_size);893    return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x);894}895 896static webgpu_encoded_op ggml_webgpu_solve_tri(webgpu_context & ctx,897                                               ggml_tensor *    src0,898                                               ggml_tensor *    src1,899                                               ggml_tensor *    dst) {900    ggml_webgpu_shader_lib_context shader_lib_ctx = {};901    shader_lib_ctx.src0                           = src0;902    shader_lib_ctx.src1                           = src1;903    shader_lib_ctx.dst                            = dst;904    shader_lib_ctx.max_wg_size        = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;905    shader_lib_ctx.wg_mem_limit_bytes = ctx->global_ctx->capabilities.limits.maxComputeWorkgroupStorageSize;906 907    webgpu_pipeline pipeline = ctx->shader_lib->get_solve_tri_pipeline(shader_lib_ctx);908 909    auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());910 911    std::vector<uint32_t> params = {912        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),913        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),914        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),915 916        (uint32_t) (src0->nb[0] / ggml_type_size(src0->type)),917        (uint32_t) (src0->nb[1] / ggml_type_size(src0->type)),918        (uint32_t) (src0->nb[2] / ggml_type_size(src0->type)),919        (uint32_t) (src0->nb[3] / ggml_type_size(src0->type)),920 921        (uint32_t) (src1->nb[0] / ggml_type_size(src1->type)),922        (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)),923        (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)),924        (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)),925 926        (uint32_t) (dst->nb[0] / ggml_type_size(dst->type)),927        (uint32_t) (dst->nb[1] / ggml_type_size(dst->type)),928        (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)),929        (uint32_t) (dst->nb[3] / ggml_type_size(dst->type)),930 931        (uint32_t) src1->ne[0],932        (uint32_t) dst->ne[2],933    };934 935    std::vector<wgpu::BindGroupEntry> entries = {936        ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0),937        ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, src1),938        ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst),939    };940 941    const uint32_t wg_x = CEIL_DIV((uint32_t) src1->ne[0], decisions->wg_size);942    const uint32_t wg_y = (uint32_t) (dst->ne[2] * dst->ne[3]);943    return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);944}945 946static webgpu_encoded_op ggml_webgpu_conv_2d(webgpu_context & ctx,947                                             ggml_tensor *    src0,948                                             ggml_tensor *    src1,949                                             ggml_tensor *    dst) {950    const int32_t s0 = ggml_get_op_params_i32(dst, 0);951    const int32_t s1 = ggml_get_op_params_i32(dst, 1);952    const int32_t p0 = ggml_get_op_params_i32(dst, 2);953    const int32_t p1 = ggml_get_op_params_i32(dst, 3);954    const int32_t d0 = ggml_get_op_params_i32(dst, 4);955    const int32_t d1 = ggml_get_op_params_i32(dst, 5);956 957    std::vector<uint32_t> params = {958        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),959        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),960        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),961 962        (uint32_t) (src0->nb[0] / ggml_type_size(src0->type)),963        (uint32_t) (src0->nb[1] / ggml_type_size(src0->type)),964        (uint32_t) (src0->nb[2] / ggml_type_size(src0->type)),965        (uint32_t) (src0->nb[3] / ggml_type_size(src0->type)),966 967        (uint32_t) (src1->nb[0] / ggml_type_size(src1->type)),968        (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)),969        (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)),970        (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)),971 972        (uint32_t) (dst->nb[0] / ggml_type_size(dst->type)),973        (uint32_t) (dst->nb[1] / ggml_type_size(dst->type)),974        (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)),975        (uint32_t) (dst->nb[3] / ggml_type_size(dst->type)),976 977        (uint32_t) src0->ne[0],978        (uint32_t) src0->ne[1],979        (uint32_t) src0->ne[2],980 981        (uint32_t) src1->ne[0],982        (uint32_t) src1->ne[1],983 984        (uint32_t) dst->ne[0],985        (uint32_t) dst->ne[1],986        (uint32_t) dst->ne[2],987        (uint32_t) dst->ne[3],988 989        (uint32_t) s0,990        (uint32_t) s1,991        (uint32_t) p0,992        (uint32_t) p1,993        (uint32_t) d0,994        (uint32_t) d1,995    };996 997    std::vector<wgpu::BindGroupEntry> entries = {998        ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0),999        ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, src1),1000        ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst),1001    };1002 1003    ggml_webgpu_shader_lib_context shader_lib_ctx = {};1004    shader_lib_ctx.src0                           = src0;1005    shader_lib_ctx.src1                           = src1;1006    shader_lib_ctx.dst                            = dst;1007    shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;1008 1009    webgpu_pipeline pipeline = ctx->shader_lib->get_conv2d_pipeline(shader_lib_ctx);1010 1011    auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());1012 1013    uint32_t wg_x;1014    uint32_t wg_y;1015    uint32_t total_wg = CEIL_DIV((uint32_t) ggml_nelements(dst), decisions->wg_size);1016    compute_2d_workgroups(total_wg, ctx->global_ctx->capabilities.limits.maxComputeWorkgroupsPerDimension, wg_x, wg_y);1017 1018    return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);1019}1020 1021// Same param/binding layout as conv_2d; the shader differs1022static webgpu_encoded_op ggml_webgpu_conv_2d_dw(webgpu_context & ctx,1023                                                ggml_tensor *    src0,1024                                                ggml_tensor *    src1,1025                                                ggml_tensor *    dst) {1026    const int32_t s0 = ggml_get_op_params_i32(dst, 0);1027    const int32_t s1 = ggml_get_op_params_i32(dst, 1);1028    const int32_t p0 = ggml_get_op_params_i32(dst, 2);1029    const int32_t p1 = ggml_get_op_params_i32(dst, 3);1030    const int32_t d0 = ggml_get_op_params_i32(dst, 4);1031    const int32_t d1 = ggml_get_op_params_i32(dst, 5);1032 1033    // Scalar params matching conv2d_dw.wgsl (weight src0 [KW,KH,1,C], input src1, output dst).1034    std::vector<uint32_t> params = {1035        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),1036        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),1037        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),1038 1039        (uint32_t) ggml_nelements(dst),1040        (uint32_t) dst->ne[2],1041        (uint32_t) dst->ne[0],1042        (uint32_t) dst->ne[1],1043        (uint32_t) src1->ne[0],1044        (uint32_t) src1->ne[1],1045        (uint32_t) src0->ne[0],1046        (uint32_t) src0->ne[1],1047 1048        (uint32_t) s0,1049        (uint32_t) s1,1050        (uint32_t) p0,1051        (uint32_t) p1,1052        (uint32_t) d0,1053        (uint32_t) d1,1054    };1055 1056    std::vector<wgpu::BindGroupEntry> entries = {1057        ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0),1058        ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, src1),1059        ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst),1060    };1061 1062    ggml_webgpu_shader_lib_context shader_lib_ctx = {};1063    shader_lib_ctx.src0                           = src0;1064    shader_lib_ctx.src1                           = src1;1065    shader_lib_ctx.dst                            = dst;1066    shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;1067 1068    // Input layout: contiguous -> WHCN, contiguous-channels -> CWHN1069    const bool      whcn      = ggml_is_contiguous(src1);1070    webgpu_pipeline pipeline  = ctx->shader_lib->get_conv2d_dw_pipeline(shader_lib_ctx, whcn);1071    auto *          decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());1072 1073    uint32_t wg_x;1074    uint32_t wg_y;1075    uint32_t total_wg = CEIL_DIV((uint32_t) ggml_nelements(dst), decisions->wg_size);1076    compute_2d_workgroups(total_wg, ctx->global_ctx->capabilities.limits.maxComputeWorkgroupsPerDimension, wg_x, wg_y);1077 1078    return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);1079}1080 1081static webgpu_encoded_op ggml_webgpu_im2col(webgpu_context & ctx,1082                                            ggml_tensor *    src0,1083                                            ggml_tensor *    src1,1084                                            ggml_tensor *    dst) {1085    const int32_t s0    = ggml_get_op_params_i32(dst, 0);1086    const int32_t s1    = ggml_get_op_params_i32(dst, 1);1087    const int32_t p0    = ggml_get_op_params_i32(dst, 2);1088    const int32_t p1    = ggml_get_op_params_i32(dst, 3);1089    const int32_t d0    = ggml_get_op_params_i32(dst, 4);1090    const int32_t d1    = ggml_get_op_params_i32(dst, 5);1091    const bool    is_2D = ggml_get_op_params_i32(dst, 6) == 1;1092 1093    const uint32_t KW = src0->ne[0];1094    const uint32_t KH = is_2D ? src0->ne[1] : 1;1095    const uint32_t IC = is_2D ? src0->ne[2] : src0->ne[1];1096 1097    const uint32_t IW = src1->ne[0];1098    const uint32_t IH = is_2D ? src1->ne[1] : 1;1099    const uint32_t N  = is_2D ? src1->ne[3] : src1->ne[2];1100 1101    const uint32_t OW = dst->ne[1];1102    const uint32_t OH = is_2D ? dst->ne[2] : 1;1103 1104    const uint32_t si0 = (uint32_t) (src1->nb[0] / ggml_type_size(src1->type));1105    const uint32_t si1 = is_2D ? (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)) : 0;1106    const uint32_t si2 = is_2D ? (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)) :1107                                 (uint32_t) (src1->nb[1] / ggml_type_size(src1->type));1108    const uint32_t si3 = is_2D ? (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)) :1109                                 (uint32_t) (src1->nb[2] / ggml_type_size(src1->type));1110 1111    const uint32_t so0 = (uint32_t) (dst->nb[0] / ggml_type_size(dst->type));1112    const uint32_t so1 = (uint32_t) (dst->nb[1] / ggml_type_size(dst->type));1113    const uint32_t so2 = is_2D ? (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)) : 0;1114    const uint32_t so3 = is_2D ? (uint32_t) (dst->nb[3] / ggml_type_size(dst->type)) :1115                                 (uint32_t) (dst->nb[2] / ggml_type_size(dst->type));1116 1117    std::vector<uint32_t> params = {1118        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),1119        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),1120 1121        si0,1122        si1,1123        si2,1124        si3,1125        so0,1126        so1,1127        so2,1128        so3,1129 1130        KW,1131        KH,1132        IC,1133 1134        IW,1135        IH,1136        N,1137 1138        OW,1139        OH,1140 1141        (uint32_t) s0,1142        (uint32_t) s1,1143        (uint32_t) p0,1144        (uint32_t) p1,1145        (uint32_t) d0,1146        (uint32_t) d1,1147    };1148 1149    std::vector<wgpu::BindGroupEntry> entries = {1150        ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src1),1151        ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, dst),1152    };1153 1154    ggml_webgpu_shader_lib_context shader_lib_ctx = {};1155    shader_lib_ctx.src0                           = src0;1156    shader_lib_ctx.src1                           = src1;1157    shader_lib_ctx.dst                            = dst;1158    shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;1159 1160    webgpu_pipeline pipeline = ctx->shader_lib->get_im2col_pipeline(shader_lib_ctx);1161 1162    auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());1163 1164    uint32_t wg_x;1165    uint32_t wg_y;1166    uint32_t total_wg = CEIL_DIV((uint32_t) ggml_nelements(dst), decisions->wg_size);1167    compute_2d_workgroups(total_wg, ctx->global_ctx->capabilities.limits.maxComputeWorkgroupsPerDimension, wg_x, wg_y);1168 1169    return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);1170}1171 1172static webgpu_encoded_op ggml_webgpu_ssm_conv(webgpu_context & ctx,1173                                              ggml_tensor *    src0,1174                                              ggml_tensor *    src1,1175                                              ggml_tensor *    dst) {1176    ggml_webgpu_shader_lib_context shader_lib_ctx = {};1177    shader_lib_ctx.src0                           = src0;1178    shader_lib_ctx.src1                           = src1;1179    shader_lib_ctx.dst                            = dst;1180    shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;1181 1182    webgpu_pipeline pipeline  = ctx->shader_lib->get_ssm_conv_pipeline(shader_lib_ctx);1183    auto *          decisions = static_cast<ggml_webgpu_ssm_conv_shader_decisions *>(pipeline.context.get());1184 1185    const uint32_t token_tiles = CEIL_DIV((uint32_t) dst->ne[1], decisions->tokens_per_wg);1186 1187    std::vector<uint32_t> params = {1188        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),1189        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),1190        (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),1191 1192        (uint32_t) (src0->nb[1] / ggml_type_size(src0->type)),1193        (uint32_t) (src0->nb[2] / ggml_type_size(src0->type)),1194        (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)),1195 1196        (uint32_t) (dst->nb[0] / ggml_type_size(dst->type)),1197        (uint32_t) (dst->nb[1] / ggml_type_size(dst->type)),1198        (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)),1199 1200        (uint32_t) src1->ne[0],

Showing the first 1,200 of 4763 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai