Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1/*2 WebGPU backend implementation.3 Note: Use ClangFormat to format this file.4*/5 6#include "ggml-webgpu.h"7 8#include "ggml-backend-impl.h"9#include "ggml-impl.h"10#include "ggml-webgpu-shader-lib.hpp"11#include "ggml.h"12 13#ifdef __EMSCRIPTEN__14# include <emscripten/emscripten.h>15#endif16 17#include <webgpu/webgpu_cpp.h>18 19#include <atomic>20#include <cstdint>21#include <cstring>22#ifdef GGML_WEBGPU_GPU_PROFILE23# include <iomanip>24#endif25#if defined(GGML_WEBGPU_DEBUG) || defined(GGML_WEBGPU_CPU_PROFILE) || defined(GGML_WEBGPU_GPU_PROFILE)26# include <iostream>27#endif28#include <memory>29#include <mutex>30#include <optional>31#include <string>32#include <utility>33#include <vector>34 35#define ROUNDUP_POW2(x, pow2) (((x) + ((pow2) - 1)) & ~((pow2) - 1))36#define CEIL_DIV(M, N) (((M) + (N) - 1) / (N))37 38// Return a rectangular grid of workgroups with minimal over-provisioned workgroups.39// Assumes that the total number of workgroups does not exceed max_per_dim^2.40static inline void compute_2d_workgroups(uint32_t total_wg, uint32_t max_per_dim, uint32_t & wg_x, uint32_t & wg_y) {41 wg_y = std::max(1u, CEIL_DIV(total_wg, max_per_dim));42 wg_x = CEIL_DIV(total_wg, wg_y);43}44 45static inline uint32_t ggml_webgpu_u32_from_f32(float value) {46 uint32_t bits;47 memcpy(&bits, &value, sizeof(bits));48 return bits;49}50 51#ifdef GGML_WEBGPU_DEBUG52# define WEBGPU_LOG_DEBUG(msg) std::cout << msg << std::endl53# define WEBGPU_DEBUG_BUF_ELEMS 51254#else55# define WEBGPU_LOG_DEBUG(msg) ((void) 0)56#endif // GGML_WEBGPU_DEBUG57 58#ifdef GGML_WEBGPU_CPU_PROFILE59// total timing (aggregated)60# define WEBGPU_CPU_PROFILE_TOTAL_START(id) auto cpu_total_start_##id = std::chrono::high_resolution_clock::now();61 62# define WEBGPU_CPU_PROFILE_TOTAL_END(id, ctx) \63 auto cpu_total_end_##id = std::chrono::high_resolution_clock::now(); \64 double cpu_total_time_##id = \65 std::chrono::duration<double, std::milli>(cpu_total_end_##id - cpu_total_start_##id).count(); \66 (ctx)->cpu_time_ms[#id] += cpu_total_time_##id;67// fine-grained timing (not included in totals)68# define WEBGPU_CPU_PROFILE_DETAIL_START(id) auto cpu_detail_start_##id = std::chrono::high_resolution_clock::now();69 70# define WEBGPU_CPU_PROFILE_DETAIL_END(id, ctx) \71 auto cpu_detail_end_##id = std::chrono::high_resolution_clock::now(); \72 double cpu_detail_time_##id = \73 std::chrono::duration<double, std::milli>(cpu_detail_end_##id - cpu_detail_start_##id).count(); \74 (ctx)->cpu_detail_ms[#id] += cpu_detail_time_##id;75#else76# define WEBGPU_CPU_PROFILE_TOTAL_START(id)77# define WEBGPU_CPU_PROFILE_TOTAL_END(id, ctx)78# define WEBGPU_CPU_PROFILE_DETAIL_START(id)79# define WEBGPU_CPU_PROFILE_DETAIL_END(id, ctx)80#endif // GGML_WEBGPU_CPU_PROFILE81 82#ifdef GGML_WEBGPU_GPU_PROFILE83# define WEBGPU_MAX_PROFILE_QUERY_COUNT 4096u84# define WEBGPU_TIMESTAMP_QUERY_BUF_SIZE_BYTES (WEBGPU_MAX_PROFILE_QUERY_COUNT * sizeof(uint64_t))85#endif86 87/* Constants */88 89#define WEBGPU_DEFAULT_COMMAND_SUBMIT_BATCH_SIZE 64u90#define WEBGPU_NUM_PARAM_SLOT_SAFETY_MARGIN 10u91#define WEBGPU_RUNTIME_WAIT_TIMEOUT_MS 30000u92#define WEBGPU_RUNTIME_WAIT_TIMEOUT_NS (WEBGPU_RUNTIME_WAIT_TIMEOUT_MS * 1e6)93#define WEBGPU_PARAMS_BUF_SIZE_BYTES 128 // enough for 32 parameters94#define WEBGPU_SET_ROWS_ERROR_BUF_SIZE_BYTES 495#define WEBGPU_STORAGE_BUF_BINDING_MULT 4 // a storage buffer binding size must be a multiple of 496 97/* End Constants */98 99// This is a "fake" base pointer, since WebGPU buffers do not have pointers to100// their locations.101static void * const webgpu_ptr_base = (void *) (uintptr_t) 0x1000; // NOLINT102 103static size_t ggml_webgpu_tensor_offset(const ggml_tensor * tensor) {104 const ggml_tensor * base_tensor = tensor->view_src ? tensor->view_src : tensor;105 return (size_t) ((uintptr_t) base_tensor->data - (uintptr_t) webgpu_ptr_base) + tensor->view_offs;106}107 108/* Struct definitions */109 110// Forward reference111static void ggml_webgpu_create_buffer(wgpu::Device & device,112 wgpu::Buffer & buffer,113 size_t size,114 wgpu::BufferUsage usage,115 const char * label);116 117// Slot-based parameter arena for compute graph encoding. Each encoded kernel118// gets a unique uniform-buffer slice within the current batch, and the slot119// cursor is reset immediately after that batch is submitted.120struct webgpu_param_arena {121 wgpu::Buffer buffer;122 size_t slot_stride = 0;123 size_t slot_size = 0;124 uint32_t slot_count = 0;125 uint32_t next_slot = 0;126 127 void init(wgpu::Device device, size_t slot_size, uint32_t slot_count, size_t alignment) {128 this->slot_stride = ROUNDUP_POW2(slot_size, alignment);129 this->slot_size = slot_size;130 this->slot_count = slot_count;131 this->next_slot = 0;132 133 ggml_webgpu_create_buffer(device, buffer, this->slot_stride * slot_count,134 wgpu::BufferUsage::CopyDst | wgpu::BufferUsage::Uniform, "ggml_webgpu_param_arena");135 }136 137 size_t alloc_slot(size_t size) {138 GGML_ASSERT(size <= slot_size);139 if (next_slot >= slot_count) {140 GGML_ABORT("ggml_webgpu: parameter arena exhausted while encoding a batch");141 }142 143 return slot_stride * next_slot++;144 }145 146 void reset() { next_slot = 0; }147 148 void cleanup() {149 if (buffer) {150 buffer.Destroy();151 buffer = nullptr;152 }153 }154 155 ~webgpu_param_arena() { this->cleanup(); }156};157 158struct webgpu_encoded_op {159 uint32_t num_kernels = 0;160#ifdef GGML_WEBGPU_GPU_PROFILE161 std::vector<std::string> pipeline_names;162#endif163};164 165struct webgpu_dispatch_desc {166 webgpu_pipeline pipeline;167 std::vector<uint32_t> params;168 std::vector<wgpu::BindGroupEntry> bind_group_entries;169 std::pair<uint32_t, uint32_t> workgroups = { 1, 1 };170};171 172struct webgpu_capabilities {173 wgpu::Limits limits;174 bool supports_subgroups = false;175 bool supports_subgroup_matrix = false;176 bool supports_dot_product = false;177 178 uint32_t sg_mat_m = 0;179 uint32_t sg_mat_n = 0;180 uint32_t sg_mat_k = 0;181 182 uint32_t subgroup_size = 0;183 uint32_t min_subgroup_size = 0;184 uint32_t max_subgroup_size = 0;185 size_t memset_bytes_per_thread;186};187 188// Stores global webgpu members189struct webgpu_global_context_struct {190 wgpu::Instance instance;191 wgpu::Adapter adapter;192 wgpu::Device device;193 wgpu::Queue queue;194 uint32_t command_submit_batch_size = WEBGPU_DEFAULT_COMMAND_SUBMIT_BATCH_SIZE;195 uint32_t max_inflight_batches = UINT32_MAX;196 197 webgpu_capabilities capabilities;198 // Shared buffer to move data from device to host199 wgpu::Buffer get_tensor_staging_buf;200 // Global mutex for get_tensor201 std::recursive_mutex mutex;202 203 wgpu::Buffer memset_params_buf;204 webgpu_pipeline memset_pipeline;205 206 std::string vendor;207 208 // TODO: We should rework the CPU profiling time handling to make it more useful. ref: https://github.com/ggml-org/llama.cpp/pull/22050209#ifdef GGML_WEBGPU_CPU_PROFILE210 // Profiling: labeled CPU time in ms (total)211 std::unordered_map<std::string, double> cpu_time_ms;212 // Profiling: detailed CPU time in ms213 std::unordered_map<std::string, double> cpu_detail_ms;214#endif215 216#ifdef GGML_WEBGPU_DEBUG217 wgpu::Buffer debug_host_buf;218 wgpu::Buffer debug_dev_buf;219#endif220 221 ~webgpu_global_context_struct() {222 if (this->get_tensor_staging_buf) {223 this->get_tensor_staging_buf.Destroy();224 this->get_tensor_staging_buf = nullptr;225 }226 if (this->memset_params_buf) {227 this->memset_params_buf.Destroy();228 this->memset_params_buf = nullptr;229 }230#ifdef GGML_WEBGPU_DEBUG231 if (this->debug_host_buf) {232 this->debug_host_buf.Destroy();233 this->debug_host_buf = nullptr;234 }235 if (this->debug_dev_buf) {236 this->debug_dev_buf.Destroy();237 this->debug_dev_buf = nullptr;238 }239#endif240 }241};242 243typedef std::shared_ptr<webgpu_global_context_struct> webgpu_global_context;244 245// All the base objects needed to run operations on a WebGPU device246struct webgpu_context_struct {247 // Points to global instances owned by ggml_backend_webgpu_reg_context248 webgpu_global_context global_ctx;249 250 std::unique_ptr<ggml_webgpu_shader_lib> shader_lib;251 252 webgpu_param_arena param_arena;253 wgpu::Buffer set_rows_dev_error_buf;254 wgpu::Buffer set_rows_host_error_buf;255 wgpu::CommandEncoder active_command_encoder;256 wgpu::ComputePassEncoder active_compute_pass;257 bool batch_compute_passes = true;258 259 size_t memset_bytes_per_thread;260 261#ifdef GGML_WEBGPU_GPU_PROFILE262 // Profiling: per-shader GPU time in ms263 std::unordered_map<std::string, double> shader_gpu_time_ms;264 wgpu::Buffer profile_timestamp_dev_buf;265 wgpu::Buffer profile_timestamp_host_buf;266 wgpu::QuerySet profile_timestamp_query_set;267 uint32_t profile_timestamp_query_count = 0;268#endif269 270 ~webgpu_context_struct() {271#ifdef GGML_WEBGPU_GPU_PROFILE272 if (this->profile_timestamp_host_buf) {273 this->profile_timestamp_host_buf.Destroy();274 this->profile_timestamp_host_buf = nullptr;275 }276 if (this->profile_timestamp_dev_buf) {277 this->profile_timestamp_dev_buf.Destroy();278 this->profile_timestamp_dev_buf = nullptr;279 }280 if (this->profile_timestamp_query_set) {281 this->profile_timestamp_query_set.Destroy();282 this->profile_timestamp_query_set = nullptr;283 }284#endif285 if (this->set_rows_host_error_buf) {286 this->set_rows_host_error_buf.Destroy();287 this->set_rows_host_error_buf = nullptr;288 }289 if (this->set_rows_dev_error_buf) {290 this->set_rows_dev_error_buf.Destroy();291 this->set_rows_dev_error_buf = nullptr;292 }293 }294};295 296typedef std::shared_ptr<webgpu_context_struct> webgpu_context;297 298// Metadata required for the ggml backend registration/discovery interface299struct ggml_backend_webgpu_reg_context {300 // Since the Instance is a global entrypoint into the WebGPU API, it lives here301 webgpu_global_context webgpu_global_ctx;302 size_t device_count;303 const char * name;304};305 306// Per-device struct for the global logical device interface307struct ggml_backend_webgpu_device_context {308 webgpu_global_context webgpu_global_ctx;309 std::string device_name;310 std::string device_desc;311};312 313// Per-thread data required to actually run WebGPU operations in a backend instance314struct ggml_backend_webgpu_context {315 webgpu_context webgpu_ctx;316 std::string name;317};318 319// Per-thread data related to buffers320struct ggml_backend_webgpu_buffer_context {321 wgpu::Buffer buffer;322 std::string label;323 webgpu_global_context global_ctx;324 325 ggml_backend_webgpu_buffer_context(wgpu::Buffer buf, std::string lbl, webgpu_global_context global_ctx_) :326 buffer(std::move(buf)),327 label(std::move(lbl)),328 global_ctx(std::move(global_ctx_)) {}329};330 331/* WebGPU object initializations */332 333static webgpu_pipeline ggml_webgpu_create_pipeline(wgpu::Device & device,334 const char * shader_code,335 const char * label,336 const std::vector<wgpu::ConstantEntry> & constants = {}) {337 wgpu::ShaderSourceWGSL shader_source;338 shader_source.code = shader_code;339 340 wgpu::ShaderModuleDescriptor shader_desc;341 shader_desc.nextInChain = &shader_source;342 343 wgpu::ShaderModule shader_module = device.CreateShaderModule(&shader_desc);344 345 wgpu::ComputePipelineDescriptor pipeline_desc;346 pipeline_desc.label = label;347 pipeline_desc.compute.module = shader_module;348 pipeline_desc.compute.entryPoint = "main"; // Entry point in the WGSL code349 pipeline_desc.layout = nullptr; // nullptr means auto layout350 if (constants.size() > 0) {351 pipeline_desc.compute.constants = constants.data();352 pipeline_desc.compute.constantCount = constants.size();353 }354 return { device.CreateComputePipeline(&pipeline_desc), label };355}356 357static void ggml_webgpu_create_buffer(wgpu::Device & device,358 wgpu::Buffer & buffer,359 size_t size,360 wgpu::BufferUsage usage,361 const char * label) {362 wgpu::BufferDescriptor buffer_desc;363 buffer_desc.size = size;364 buffer_desc.usage = usage;365 buffer_desc.label = label;366 buffer_desc.mappedAtCreation = false;367 368 // TODO: error handling369 buffer = device.CreateBuffer(&buffer_desc);370}371 372static wgpu::Buffer ggml_webgpu_tensor_buf(const ggml_tensor * tensor) {373 ggml_backend_webgpu_buffer_context * ctx = (ggml_backend_webgpu_buffer_context *) tensor->buffer->context;374 return ctx->buffer;375}376 377static size_t ggml_webgpu_tensor_misalignment(const ggml_tensor * t, size_t alignment) {378 size_t offset = ggml_webgpu_tensor_offset(t);379 return offset & (alignment - 1);380}381 382static size_t ggml_webgpu_tensor_misalignment(webgpu_context & ctx, const ggml_tensor * t) {383 return ggml_webgpu_tensor_misalignment(t, ctx->global_ctx->capabilities.limits.minStorageBufferOffsetAlignment);384}385 386static size_t ggml_webgpu_tensor_align_offset(const ggml_tensor * t, size_t alignment) {387 size_t offset = ggml_webgpu_tensor_offset(t);388 return offset & ~(alignment - 1);389}390 391static size_t ggml_webgpu_tensor_align_offset(webgpu_context & ctx, const ggml_tensor * t) {392 return ggml_webgpu_tensor_align_offset(t, ctx->global_ctx->capabilities.limits.minStorageBufferOffsetAlignment);393}394 395static size_t ggml_webgpu_tensor_binding_size(const ggml_tensor * t, size_t alignment) {396 return ROUNDUP_POW2(ggml_nbytes(t) + ggml_webgpu_tensor_misalignment(t, alignment),397 WEBGPU_STORAGE_BUF_BINDING_MULT);398}399 400static size_t ggml_webgpu_tensor_binding_size(webgpu_context & ctx, const ggml_tensor * t) {401 return ggml_webgpu_tensor_binding_size(t, ctx->global_ctx->capabilities.limits.minStorageBufferOffsetAlignment);402}403 404static bool ggml_webgpu_tensor_binding_overlap(const webgpu_global_context & global_ctx,405 const ggml_tensor * a,406 const ggml_tensor * b) {407 if (a->buffer != b->buffer) {408 return false;409 }410 411 const size_t alignment = global_ctx->capabilities.limits.minStorageBufferOffsetAlignment;412 const size_t a_offset = ggml_webgpu_tensor_align_offset(a, alignment);413 const size_t b_offset = ggml_webgpu_tensor_align_offset(b, alignment);414 return a_offset < b_offset + ggml_webgpu_tensor_binding_size(b, alignment) &&415 b_offset < a_offset + ggml_webgpu_tensor_binding_size(a, alignment);416}417 418static bool ggml_webgpu_tensor_binding_overlap_range(const webgpu_global_context & global_ctx,419 ggml_tensor * tensor,420 ggml_backend_buffer_t buffer,421 size_t offset,422 size_t size) {423 if (tensor->buffer != buffer) {424 return false;425 }426 427 const size_t alignment = global_ctx->capabilities.limits.minStorageBufferOffsetAlignment;428 const size_t tensor_offset = ggml_webgpu_tensor_align_offset(tensor, alignment);429 return tensor_offset < offset + size && offset < tensor_offset + ggml_webgpu_tensor_binding_size(tensor, alignment);430}431 432struct ggml_webgpu_merged_binding_range {433 size_t offset;434 size_t size;435};436 437static ggml_webgpu_merged_binding_range ggml_webgpu_tensor_merged_binding_range(438 webgpu_context & ctx,439 std::initializer_list<ggml_tensor *> tensors) {440 size_t merged_offset = SIZE_MAX;441 size_t merged_end = 0;442 443 for (ggml_tensor * tensor : tensors) {444 const size_t bind_offset = ggml_webgpu_tensor_align_offset(ctx, tensor);445 const size_t bind_end = bind_offset + ggml_webgpu_tensor_binding_size(ctx, tensor);446 447 merged_offset = std::min(merged_offset, bind_offset);448 merged_end = std::max(merged_end, bind_end);449 }450 451 return { merged_offset, merged_end - merged_offset };452}453 454static uint32_t ggml_webgpu_tensor_merged_element_offset(const ggml_tensor * tensor,455 const ggml_webgpu_merged_binding_range & merged_range) {456 return (uint32_t) ((ggml_webgpu_tensor_offset(tensor) - merged_range.offset) / ggml_type_size(tensor->type));457}458 459static wgpu::BindGroupEntry ggml_webgpu_make_bind_group_entry(uint32_t binding,460 wgpu::Buffer buffer,461 uint64_t offset,462 uint64_t size) {463 wgpu::BindGroupEntry entry = {};464 entry.binding = binding;465 entry.buffer = std::move(buffer);466 entry.offset = offset;467 entry.size = size;468 return entry;469}470 471static wgpu::BindGroupEntry ggml_webgpu_make_tensor_bind_group_entry(webgpu_context & ctx,472 uint32_t binding,473 ggml_tensor * tensor) {474 return ggml_webgpu_make_bind_group_entry(binding, ggml_webgpu_tensor_buf(tensor),475 ggml_webgpu_tensor_align_offset(ctx, tensor),476 ggml_webgpu_tensor_binding_size(ctx, tensor));477}478 479/** End WebGPU object initializations */480 481/** WebGPU Actions */482 483template <typename T>484static void ggml_backend_webgpu_check_wait_status(wgpu::WaitStatus wait_status,485 T callback_status,486 T success_status,487 const char * wait_name,488 const char * failure_name,489 const char * callback_message) {490 if (wait_status == wgpu::WaitStatus::TimedOut) {491 GGML_ABORT("ggml_webgpu: %s timed out after %u ms\n", wait_name, WEBGPU_RUNTIME_WAIT_TIMEOUT_MS);492 }493 if (wait_status == wgpu::WaitStatus::Error) {494 GGML_ABORT("ggml_webgpu: %s failed\n", wait_name);495 }496 if (callback_status != success_status) {497 GGML_ABORT("ggml_webgpu: %s failed with status %d: %s\n", failure_name, static_cast<int>(callback_status),498 callback_message);499 }500}501 502// TODO: these next two functions may want tuning across different platforms and workloads,503static uint32_t ggml_backend_webgpu_get_max_inflight_batches() {504 return UINT32_MAX;505}506 507static uint32_t ggml_backend_webgpu_get_command_submit_batch_size() {508 return WEBGPU_DEFAULT_COMMAND_SUBMIT_BATCH_SIZE;509}510 511static void ggml_backend_webgpu_wait_queue(webgpu_global_context & ctx) {512 wgpu::QueueWorkDoneStatus callback_status = wgpu::QueueWorkDoneStatus::Error;513 std::string callback_message;514 515 const wgpu::WaitStatus wait_status = ctx->instance.WaitAny(516 ctx->queue.OnSubmittedWorkDone(517 wgpu::CallbackMode::AllowSpontaneous,518 [&callback_status, &callback_message](wgpu::QueueWorkDoneStatus status, wgpu::StringView message) {519 callback_status = status;520 callback_message = std::string(message);521 }),522 WEBGPU_RUNTIME_WAIT_TIMEOUT_NS);523 524 ggml_backend_webgpu_check_wait_status(wait_status, callback_status, wgpu::QueueWorkDoneStatus::Success,525 "Queue wait", "Queue work", callback_message.c_str());526}527 528static void ggml_backend_webgpu_map_buffer(webgpu_global_context & ctx,529 wgpu::Buffer & buffer,530 wgpu::MapMode mode,531 size_t offset,532 size_t size) {533 wgpu::MapAsyncStatus callback_status = wgpu::MapAsyncStatus::Error;534 std::string callback_message;535 536 const wgpu::WaitStatus wait_status = ctx->instance.WaitAny(537 buffer.MapAsync(mode, offset, size, wgpu::CallbackMode::AllowSpontaneous,538 [&callback_status, &callback_message](wgpu::MapAsyncStatus status, wgpu::StringView message) {539 callback_status = status;540 callback_message = std::string(message);541 }),542 WEBGPU_RUNTIME_WAIT_TIMEOUT_NS);543 544 ggml_backend_webgpu_check_wait_status(wait_status, callback_status, wgpu::MapAsyncStatus::Success,545 "Buffer map wait", "Buffer map", callback_message.c_str());546}547 548static void ggml_backend_webgpu_submit_commands(webgpu_context & ctx,549 const wgpu::CommandBuffer commands,550 uint32_t & num_inflight_batches) {551 if (num_inflight_batches >= ctx->global_ctx->max_inflight_batches) {552 ggml_backend_webgpu_wait_queue(ctx->global_ctx);553 num_inflight_batches = 0;554 }555 556 ctx->global_ctx->queue.Submit(1, &commands);557 num_inflight_batches++;558}559 560#ifdef GGML_WEBGPU_DEBUG561// This function adds debugging information to shaders, as WebGPU does not support printing directly.562// To use, add a bind group entry to the setup for the shader you are debugging, add the buffer and563// debug statements in the shader, and then call this function after encoding the commands and submitting them.564static void ggml_backend_webgpu_debug(webgpu_global_context & ctx) {565 wgpu::CommandEncoder encoder = ctx->device.CreateCommandEncoder();566 encoder.CopyBufferToBuffer(ctx->debug_dev_buf, 0, ctx->debug_host_buf, 0, ctx->debug_host_buf.GetSize());567 wgpu::CommandBuffer commands = encoder.Finish();568 ctx->queue.Submit(1, &commands);569 ggml_backend_webgpu_map_buffer(ctx, ctx->debug_host_buf, wgpu::MapMode::Read, 0, ctx->debug_host_buf.GetSize());570 const float * debug_data = (const float *) ctx->debug_host_buf.GetConstMappedRange();571 std::cout << "debug[0]: " << debug_data[0] << "\n";572 ctx->debug_host_buf.Unmap();573}574#endif575 576static webgpu_encoded_op ggml_backend_webgpu_build_multi(webgpu_context & ctx,577 const std::vector<webgpu_dispatch_desc> & dispatches) {578 webgpu_encoded_op result = {};579 std::vector<wgpu::BindGroup> bind_groups;580 std::vector<size_t> param_offsets;581 result.num_kernels = dispatches.size();582 583 for (size_t i = 0; i < dispatches.size(); i++) {584 const webgpu_dispatch_desc & dispatch = dispatches[i];585 const size_t param_size = dispatch.params.size() * sizeof(uint32_t);586 const size_t param_offset = ctx->param_arena.alloc_slot(param_size);587 588 std::vector<wgpu::BindGroupEntry> entries = dispatch.bind_group_entries;589 uint32_t params_binding_num = entries.size();590 entries.push_back(ggml_webgpu_make_bind_group_entry(params_binding_num, ctx->param_arena.buffer, param_offset,591 ctx->param_arena.slot_size));592 593 wgpu::BindGroupDescriptor bind_group_desc;594 bind_group_desc.layout = dispatch.pipeline.pipeline.GetBindGroupLayout(0);595 bind_group_desc.entryCount = entries.size();596 bind_group_desc.entries = entries.data();597 bind_group_desc.label = dispatch.pipeline.name.c_str();598 bind_groups.push_back(ctx->global_ctx->device.CreateBindGroup(&bind_group_desc));599 param_offsets.push_back(param_offset);600 }601 602 for (size_t i = 0; i < param_offsets.size(); i++) {603 ctx->global_ctx->queue.WriteBuffer(ctx->param_arena.buffer, param_offsets[i], dispatches[i].params.data(),604 dispatches[i].params.size() * sizeof(uint32_t));605 }606 607#ifdef GGML_WEBGPU_GPU_PROFILE608 for (size_t i = 0; i < dispatches.size(); i++) {609 GGML_ASSERT(ctx->profile_timestamp_query_count + 2 <= WEBGPU_MAX_PROFILE_QUERY_COUNT);610 const uint32_t query_begin = ctx->profile_timestamp_query_count++;611 const uint32_t query_end = ctx->profile_timestamp_query_count++;612 613 wgpu::PassTimestampWrites ts_writes = {};614 ts_writes.querySet = ctx->profile_timestamp_query_set;615 ts_writes.beginningOfPassWriteIndex = query_begin;616 ts_writes.endOfPassWriteIndex = query_end;617 wgpu::ComputePassDescriptor pass_desc = {};618 pass_desc.timestampWrites = &ts_writes;619 620 wgpu::ComputePassEncoder pass = ctx->active_command_encoder.BeginComputePass(&pass_desc);621 622 pass.SetPipeline(dispatches[i].pipeline.pipeline);623 pass.SetBindGroup(0, bind_groups[i]);624 pass.DispatchWorkgroups(dispatches[i].workgroups.first, dispatches[i].workgroups.second, 1);625 pass.End();626 result.pipeline_names.push_back(dispatches[i].pipeline.name);627 }628#else629 for (size_t i = 0; i < dispatches.size(); i++) {630 if (ctx->batch_compute_passes) {631 ctx->active_compute_pass.SetPipeline(dispatches[i].pipeline.pipeline);632 ctx->active_compute_pass.SetBindGroup(0, bind_groups[i]);633 ctx->active_compute_pass.DispatchWorkgroups(dispatches[i].workgroups.first, dispatches[i].workgroups.second,634 1);635 } else {636 wgpu::ComputePassEncoder pass = ctx->active_command_encoder.BeginComputePass();637 pass.SetPipeline(dispatches[i].pipeline.pipeline);638 pass.SetBindGroup(0, bind_groups[i]);639 pass.DispatchWorkgroups(dispatches[i].workgroups.first, dispatches[i].workgroups.second, 1);640 pass.End();641 }642 }643#endif644 645 return result;646}647 648static webgpu_encoded_op ggml_backend_webgpu_build(webgpu_context & ctx,649 webgpu_pipeline & pipeline,650 std::vector<uint32_t> params,651 std::vector<wgpu::BindGroupEntry> bind_group_entries,652 uint32_t wg_x,653 uint32_t wg_y = 1) {654 return ggml_backend_webgpu_build_multi(655 ctx, {656 { pipeline, std::move(params), std::move(bind_group_entries), { wg_x, wg_y } },657 });658}659 660static void ggml_backend_webgpu_buffer_memset(webgpu_global_context & ctx,661 wgpu::Buffer & buf,662 uint32_t value,663 size_t offset,664 size_t size) {665 std::vector<uint32_t> params = { (uint32_t) offset, (uint32_t) size, value };666 std::vector<wgpu::BindGroupEntry> entries = { ggml_webgpu_make_bind_group_entry(0, buf, 0, buf.GetSize()) };667 size_t bytes_per_wg =668 ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup * ctx->capabilities.memset_bytes_per_thread;669 uint32_t wg_x = CEIL_DIV(size + 3, bytes_per_wg);670 671 ctx->queue.WriteBuffer(ctx->memset_params_buf, 0, params.data(), params.size() * sizeof(uint32_t));672 673 wgpu::BindGroupEntry params_entry = {};674 params_entry.binding = 1;675 params_entry.buffer = ctx->memset_params_buf;676 params_entry.offset = 0;677 params_entry.size = WEBGPU_PARAMS_BUF_SIZE_BYTES;678 entries.push_back(params_entry);679 680 wgpu::BindGroupDescriptor bind_group_desc;681 bind_group_desc.layout = ctx->memset_pipeline.pipeline.GetBindGroupLayout(0);682 bind_group_desc.entryCount = entries.size();683 bind_group_desc.entries = entries.data();684 bind_group_desc.label = ctx->memset_pipeline.name.c_str();685 wgpu::BindGroup bind_group = ctx->device.CreateBindGroup(&bind_group_desc);686 687 wgpu::CommandEncoder encoder = ctx->device.CreateCommandEncoder();688 wgpu::ComputePassEncoder pass = encoder.BeginComputePass();689 pass.SetPipeline(ctx->memset_pipeline.pipeline);690 pass.SetBindGroup(0, bind_group);691 pass.DispatchWorkgroups(wg_x, 1, 1);692 pass.End();693 694 wgpu::CommandBuffer command = encoder.Finish();695 std::vector<wgpu::CommandBuffer> commands = { command };696 ctx->queue.Submit(commands.size(), commands.data());697}698 699/** End WebGPU Actions */700 701/** GGML Backend Interface */702 703static const char * ggml_backend_webgpu_name(ggml_backend_t backend) {704 ggml_backend_webgpu_context * ctx = (ggml_backend_webgpu_context *) backend->context;705 return ctx->name.c_str();706}707 708static void ggml_backend_webgpu_free(ggml_backend_t backend) {709 ggml_backend_webgpu_context * ctx = (ggml_backend_webgpu_context *) backend->context;710 WEBGPU_LOG_DEBUG("ggml_backend_webgpu_free(" << ctx->name << ")");711 712#ifdef GGML_WEBGPU_CPU_PROFILE713 std::cout << "\n[ggml_webgpu cpu profiling summary]\n";714 double total_cpu = 0.0;715 for (const auto & kv : ctx->webgpu_ctx->global_ctx->cpu_time_ms) {716 total_cpu += kv.second;717 }718 std::cout << "ggml_webgpu: total cpu time: " << total_cpu << " ms\n";719 std::cout << "ggml_webgpu: cpu breakdown:\n";720 for (const auto & kv : ctx->webgpu_ctx->global_ctx->cpu_time_ms) {721 double pct = (total_cpu > 0.0) ? (kv.second / total_cpu * 100.0) : 0.0;722 std::cout << "ggml_webgpu: " << kv.first << ": " << kv.second << " ms (" << pct << "%)\n";723 }724 if (ctx->webgpu_ctx->global_ctx->cpu_detail_ms.size() > 0) {725 std::cout << "ggml_webgpu: cpu detailed breakdown:\n";726 }727 for (const auto & kv : ctx->webgpu_ctx->global_ctx->cpu_detail_ms) {728 double pct = (total_cpu > 0.0) ? (kv.second / total_cpu * 100.0) : 0.0;729 std::cout << "ggml_webgpu: " << kv.first << ": " << kv.second << " ms (" << pct << "%)\n";730 }731#endif732 733#ifdef GGML_WEBGPU_GPU_PROFILE734 std::cout << "\n[ggml_webgpu gpu profiling summary]\n";735 double total_gpu = 0.0;736 for (const auto & kv : ctx->webgpu_ctx->shader_gpu_time_ms) {737 total_gpu += kv.second;738 }739 std::cout << "ggml_webgpu: total gpu time (all shaders): " << total_gpu << " ms\n";740 std::cout << "\nggml_webgpu: gpu breakdown:\n";741 for (const auto & kv : ctx->webgpu_ctx->shader_gpu_time_ms) {742 double pct = (total_gpu > 0.0) ? (kv.second / total_gpu * 100.0) : 0.0;743 std::cout << "ggml_webgpu: " << kv.first << ": " << kv.second << " ms (" << std::fixed << std::setprecision(2)744 << pct << "%)\n";745 }746#endif747 748#if defined(GGML_WEBGPU_CPU_PROFILE) && defined(GGML_WEBGPU_GPU_PROFILE)749 std::cout << "ggml_webgpu: gpu/cpu ratio: " << (total_cpu > 0.0 ? total_gpu / total_cpu : 0.0) << "\n";750#endif751 752 delete ctx;753 delete backend;754}755 756static webgpu_encoded_op ggml_webgpu_cpy(webgpu_context & ctx, ggml_tensor * src, ggml_tensor * dst) {757 ggml_webgpu_shader_lib_context shader_lib_ctx = {};758 shader_lib_ctx.src0 = src;759 shader_lib_ctx.dst = dst;760 shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;761 762 webgpu_pipeline pipeline = ctx->shader_lib->get_cpy_pipeline(shader_lib_ctx);763 764 auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());765 766 uint32_t ne = (uint32_t) ggml_nelements(dst);767 768 std::vector<uint32_t> params = {769 ne, (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src) / ggml_type_size(src->type)),770 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),771 // Convert byte-strides to element-strides772 (uint32_t) (src->nb[0] / ggml_type_size(src->type)), (uint32_t) (src->nb[1] / ggml_type_size(src->type)),773 (uint32_t) (src->nb[2] / ggml_type_size(src->type)), (uint32_t) (src->nb[3] / ggml_type_size(src->type)),774 (uint32_t) (dst->nb[0] / ggml_type_size(dst->type)), (uint32_t) (dst->nb[1] / ggml_type_size(dst->type)),775 (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)), (uint32_t) (dst->nb[3] / ggml_type_size(dst->type)),776 // Logical shapes777 (uint32_t) src->ne[0], (uint32_t) src->ne[1], (uint32_t) src->ne[2], (uint32_t) dst->ne[0],778 (uint32_t) dst->ne[1], (uint32_t) dst->ne[2]779 };780 781 std::vector<wgpu::BindGroupEntry> entries = {782 ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src),783 ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, dst),784 };785 786 uint32_t wg_x;787 uint32_t wg_y;788 uint32_t total_wg = CEIL_DIV(ne, decisions->wg_size);789 compute_2d_workgroups(total_wg, ctx->global_ctx->capabilities.limits.maxComputeWorkgroupsPerDimension, wg_x, wg_y);790 return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);791}792 793static webgpu_encoded_op ggml_webgpu_set(webgpu_context & ctx,794 ggml_tensor * src0,795 ggml_tensor * src1,796 ggml_tensor * dst) {797 ggml_webgpu_shader_lib_context shader_lib_ctx = {};798 shader_lib_ctx.src0 = src0;799 shader_lib_ctx.src1 = src1;800 shader_lib_ctx.dst = dst;801 shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;802 803 webgpu_pipeline pipeline = ctx->shader_lib->get_set_pipeline(shader_lib_ctx);804 805 auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());806 const bool inplace = decisions->inplace;807 808 const uint32_t ne = inplace ? (uint32_t) ggml_nelements(src1) : (uint32_t) ggml_nelements(dst);809 const uint32_t dst_type_size = (uint32_t) ggml_type_size(dst->type);810 811 std::vector<uint32_t> params = {812 ne,813 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),814 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),815 (uint32_t) (((const int32_t *) dst->op_params)[3] / dst_type_size),816 817 (uint32_t) (src1->nb[0] / ggml_type_size(src1->type)),818 (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)),819 (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)),820 (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)),821 822 1u,823 (uint32_t) (((const int32_t *) dst->op_params)[0] / dst_type_size),824 (uint32_t) (((const int32_t *) dst->op_params)[1] / dst_type_size),825 (uint32_t) (((const int32_t *) dst->op_params)[2] / dst_type_size),826 827 (uint32_t) src1->ne[0],828 (uint32_t) src1->ne[1],829 (uint32_t) src1->ne[2],830 (uint32_t) src1->ne[3],831 };832 833 std::vector<wgpu::BindGroupEntry> entries;834 uint32_t binding_index = 0;835 if (!inplace) {836 entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0));837 binding_index++;838 }839 entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, binding_index, src1));840 entries.push_back(ggml_webgpu_make_tensor_bind_group_entry(ctx, binding_index + 1, dst));841 842 uint32_t wg_x = CEIL_DIV(ne, decisions->wg_size);843 return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x);844}845 846static webgpu_encoded_op ggml_webgpu_pad(webgpu_context & ctx, ggml_tensor * src, ggml_tensor * dst) {847 ggml_webgpu_shader_lib_context shader_lib_ctx = {};848 shader_lib_ctx.src0 = src;849 shader_lib_ctx.dst = dst;850 shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;851 852 webgpu_pipeline pipeline = ctx->shader_lib->get_pad_pipeline(shader_lib_ctx);853 854 auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());855 856 const uint32_t ne = (uint32_t) ggml_nelements(dst);857 858 std::vector<uint32_t> params = {859 ne,860 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src) / ggml_type_size(src->type)),861 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),862 // Strides (in elements)863 (uint32_t) (src->nb[0] / ggml_type_size(src->type)),864 (uint32_t) (src->nb[1] / ggml_type_size(src->type)),865 (uint32_t) (src->nb[2] / ggml_type_size(src->type)),866 (uint32_t) (src->nb[3] / ggml_type_size(src->type)),867 // Shapes868 (uint32_t) src->ne[0],869 (uint32_t) src->ne[1],870 (uint32_t) src->ne[2],871 (uint32_t) src->ne[3],872 (uint32_t) dst->ne[0],873 (uint32_t) dst->ne[1],874 (uint32_t) dst->ne[2],875 (uint32_t) dst->ne[3],876 // Pad sizes877 (uint32_t) ggml_get_op_params_i32(dst, 0),878 (uint32_t) ggml_get_op_params_i32(dst, 1),879 (uint32_t) ggml_get_op_params_i32(dst, 2),880 (uint32_t) ggml_get_op_params_i32(dst, 3),881 (uint32_t) ggml_get_op_params_i32(dst, 4),882 (uint32_t) ggml_get_op_params_i32(dst, 5),883 (uint32_t) ggml_get_op_params_i32(dst, 6),884 (uint32_t) ggml_get_op_params_i32(dst, 7),885 };886 887 std::vector<wgpu::BindGroupEntry> entries = {888 ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src),889 ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, dst),890 };891 892 uint32_t wg_x = CEIL_DIV(ne, decisions->wg_size);893 return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x);894}895 896static webgpu_encoded_op ggml_webgpu_solve_tri(webgpu_context & ctx,897 ggml_tensor * src0,898 ggml_tensor * src1,899 ggml_tensor * dst) {900 ggml_webgpu_shader_lib_context shader_lib_ctx = {};901 shader_lib_ctx.src0 = src0;902 shader_lib_ctx.src1 = src1;903 shader_lib_ctx.dst = dst;904 shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;905 shader_lib_ctx.wg_mem_limit_bytes = ctx->global_ctx->capabilities.limits.maxComputeWorkgroupStorageSize;906 907 webgpu_pipeline pipeline = ctx->shader_lib->get_solve_tri_pipeline(shader_lib_ctx);908 909 auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());910 911 std::vector<uint32_t> params = {912 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),913 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),914 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),915 916 (uint32_t) (src0->nb[0] / ggml_type_size(src0->type)),917 (uint32_t) (src0->nb[1] / ggml_type_size(src0->type)),918 (uint32_t) (src0->nb[2] / ggml_type_size(src0->type)),919 (uint32_t) (src0->nb[3] / ggml_type_size(src0->type)),920 921 (uint32_t) (src1->nb[0] / ggml_type_size(src1->type)),922 (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)),923 (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)),924 (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)),925 926 (uint32_t) (dst->nb[0] / ggml_type_size(dst->type)),927 (uint32_t) (dst->nb[1] / ggml_type_size(dst->type)),928 (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)),929 (uint32_t) (dst->nb[3] / ggml_type_size(dst->type)),930 931 (uint32_t) src1->ne[0],932 (uint32_t) dst->ne[2],933 };934 935 std::vector<wgpu::BindGroupEntry> entries = {936 ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0),937 ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, src1),938 ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst),939 };940 941 const uint32_t wg_x = CEIL_DIV((uint32_t) src1->ne[0], decisions->wg_size);942 const uint32_t wg_y = (uint32_t) (dst->ne[2] * dst->ne[3]);943 return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);944}945 946static webgpu_encoded_op ggml_webgpu_conv_2d(webgpu_context & ctx,947 ggml_tensor * src0,948 ggml_tensor * src1,949 ggml_tensor * dst) {950 const int32_t s0 = ggml_get_op_params_i32(dst, 0);951 const int32_t s1 = ggml_get_op_params_i32(dst, 1);952 const int32_t p0 = ggml_get_op_params_i32(dst, 2);953 const int32_t p1 = ggml_get_op_params_i32(dst, 3);954 const int32_t d0 = ggml_get_op_params_i32(dst, 4);955 const int32_t d1 = ggml_get_op_params_i32(dst, 5);956 957 std::vector<uint32_t> params = {958 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),959 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),960 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),961 962 (uint32_t) (src0->nb[0] / ggml_type_size(src0->type)),963 (uint32_t) (src0->nb[1] / ggml_type_size(src0->type)),964 (uint32_t) (src0->nb[2] / ggml_type_size(src0->type)),965 (uint32_t) (src0->nb[3] / ggml_type_size(src0->type)),966 967 (uint32_t) (src1->nb[0] / ggml_type_size(src1->type)),968 (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)),969 (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)),970 (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)),971 972 (uint32_t) (dst->nb[0] / ggml_type_size(dst->type)),973 (uint32_t) (dst->nb[1] / ggml_type_size(dst->type)),974 (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)),975 (uint32_t) (dst->nb[3] / ggml_type_size(dst->type)),976 977 (uint32_t) src0->ne[0],978 (uint32_t) src0->ne[1],979 (uint32_t) src0->ne[2],980 981 (uint32_t) src1->ne[0],982 (uint32_t) src1->ne[1],983 984 (uint32_t) dst->ne[0],985 (uint32_t) dst->ne[1],986 (uint32_t) dst->ne[2],987 (uint32_t) dst->ne[3],988 989 (uint32_t) s0,990 (uint32_t) s1,991 (uint32_t) p0,992 (uint32_t) p1,993 (uint32_t) d0,994 (uint32_t) d1,995 };996 997 std::vector<wgpu::BindGroupEntry> entries = {998 ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0),999 ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, src1),1000 ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst),1001 };1002 1003 ggml_webgpu_shader_lib_context shader_lib_ctx = {};1004 shader_lib_ctx.src0 = src0;1005 shader_lib_ctx.src1 = src1;1006 shader_lib_ctx.dst = dst;1007 shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;1008 1009 webgpu_pipeline pipeline = ctx->shader_lib->get_conv2d_pipeline(shader_lib_ctx);1010 1011 auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());1012 1013 uint32_t wg_x;1014 uint32_t wg_y;1015 uint32_t total_wg = CEIL_DIV((uint32_t) ggml_nelements(dst), decisions->wg_size);1016 compute_2d_workgroups(total_wg, ctx->global_ctx->capabilities.limits.maxComputeWorkgroupsPerDimension, wg_x, wg_y);1017 1018 return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);1019}1020 1021// Same param/binding layout as conv_2d; the shader differs1022static webgpu_encoded_op ggml_webgpu_conv_2d_dw(webgpu_context & ctx,1023 ggml_tensor * src0,1024 ggml_tensor * src1,1025 ggml_tensor * dst) {1026 const int32_t s0 = ggml_get_op_params_i32(dst, 0);1027 const int32_t s1 = ggml_get_op_params_i32(dst, 1);1028 const int32_t p0 = ggml_get_op_params_i32(dst, 2);1029 const int32_t p1 = ggml_get_op_params_i32(dst, 3);1030 const int32_t d0 = ggml_get_op_params_i32(dst, 4);1031 const int32_t d1 = ggml_get_op_params_i32(dst, 5);1032 1033 // Scalar params matching conv2d_dw.wgsl (weight src0 [KW,KH,1,C], input src1, output dst).1034 std::vector<uint32_t> params = {1035 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),1036 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),1037 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),1038 1039 (uint32_t) ggml_nelements(dst),1040 (uint32_t) dst->ne[2],1041 (uint32_t) dst->ne[0],1042 (uint32_t) dst->ne[1],1043 (uint32_t) src1->ne[0],1044 (uint32_t) src1->ne[1],1045 (uint32_t) src0->ne[0],1046 (uint32_t) src0->ne[1],1047 1048 (uint32_t) s0,1049 (uint32_t) s1,1050 (uint32_t) p0,1051 (uint32_t) p1,1052 (uint32_t) d0,1053 (uint32_t) d1,1054 };1055 1056 std::vector<wgpu::BindGroupEntry> entries = {1057 ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src0),1058 ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, src1),1059 ggml_webgpu_make_tensor_bind_group_entry(ctx, 2, dst),1060 };1061 1062 ggml_webgpu_shader_lib_context shader_lib_ctx = {};1063 shader_lib_ctx.src0 = src0;1064 shader_lib_ctx.src1 = src1;1065 shader_lib_ctx.dst = dst;1066 shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;1067 1068 // Input layout: contiguous -> WHCN, contiguous-channels -> CWHN1069 const bool whcn = ggml_is_contiguous(src1);1070 webgpu_pipeline pipeline = ctx->shader_lib->get_conv2d_dw_pipeline(shader_lib_ctx, whcn);1071 auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());1072 1073 uint32_t wg_x;1074 uint32_t wg_y;1075 uint32_t total_wg = CEIL_DIV((uint32_t) ggml_nelements(dst), decisions->wg_size);1076 compute_2d_workgroups(total_wg, ctx->global_ctx->capabilities.limits.maxComputeWorkgroupsPerDimension, wg_x, wg_y);1077 1078 return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);1079}1080 1081static webgpu_encoded_op ggml_webgpu_im2col(webgpu_context & ctx,1082 ggml_tensor * src0,1083 ggml_tensor * src1,1084 ggml_tensor * dst) {1085 const int32_t s0 = ggml_get_op_params_i32(dst, 0);1086 const int32_t s1 = ggml_get_op_params_i32(dst, 1);1087 const int32_t p0 = ggml_get_op_params_i32(dst, 2);1088 const int32_t p1 = ggml_get_op_params_i32(dst, 3);1089 const int32_t d0 = ggml_get_op_params_i32(dst, 4);1090 const int32_t d1 = ggml_get_op_params_i32(dst, 5);1091 const bool is_2D = ggml_get_op_params_i32(dst, 6) == 1;1092 1093 const uint32_t KW = src0->ne[0];1094 const uint32_t KH = is_2D ? src0->ne[1] : 1;1095 const uint32_t IC = is_2D ? src0->ne[2] : src0->ne[1];1096 1097 const uint32_t IW = src1->ne[0];1098 const uint32_t IH = is_2D ? src1->ne[1] : 1;1099 const uint32_t N = is_2D ? src1->ne[3] : src1->ne[2];1100 1101 const uint32_t OW = dst->ne[1];1102 const uint32_t OH = is_2D ? dst->ne[2] : 1;1103 1104 const uint32_t si0 = (uint32_t) (src1->nb[0] / ggml_type_size(src1->type));1105 const uint32_t si1 = is_2D ? (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)) : 0;1106 const uint32_t si2 = is_2D ? (uint32_t) (src1->nb[2] / ggml_type_size(src1->type)) :1107 (uint32_t) (src1->nb[1] / ggml_type_size(src1->type));1108 const uint32_t si3 = is_2D ? (uint32_t) (src1->nb[3] / ggml_type_size(src1->type)) :1109 (uint32_t) (src1->nb[2] / ggml_type_size(src1->type));1110 1111 const uint32_t so0 = (uint32_t) (dst->nb[0] / ggml_type_size(dst->type));1112 const uint32_t so1 = (uint32_t) (dst->nb[1] / ggml_type_size(dst->type));1113 const uint32_t so2 = is_2D ? (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)) : 0;1114 const uint32_t so3 = is_2D ? (uint32_t) (dst->nb[3] / ggml_type_size(dst->type)) :1115 (uint32_t) (dst->nb[2] / ggml_type_size(dst->type));1116 1117 std::vector<uint32_t> params = {1118 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),1119 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),1120 1121 si0,1122 si1,1123 si2,1124 si3,1125 so0,1126 so1,1127 so2,1128 so3,1129 1130 KW,1131 KH,1132 IC,1133 1134 IW,1135 IH,1136 N,1137 1138 OW,1139 OH,1140 1141 (uint32_t) s0,1142 (uint32_t) s1,1143 (uint32_t) p0,1144 (uint32_t) p1,1145 (uint32_t) d0,1146 (uint32_t) d1,1147 };1148 1149 std::vector<wgpu::BindGroupEntry> entries = {1150 ggml_webgpu_make_tensor_bind_group_entry(ctx, 0, src1),1151 ggml_webgpu_make_tensor_bind_group_entry(ctx, 1, dst),1152 };1153 1154 ggml_webgpu_shader_lib_context shader_lib_ctx = {};1155 shader_lib_ctx.src0 = src0;1156 shader_lib_ctx.src1 = src1;1157 shader_lib_ctx.dst = dst;1158 shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;1159 1160 webgpu_pipeline pipeline = ctx->shader_lib->get_im2col_pipeline(shader_lib_ctx);1161 1162 auto * decisions = static_cast<ggml_webgpu_generic_shader_decisions *>(pipeline.context.get());1163 1164 uint32_t wg_x;1165 uint32_t wg_y;1166 uint32_t total_wg = CEIL_DIV((uint32_t) ggml_nelements(dst), decisions->wg_size);1167 compute_2d_workgroups(total_wg, ctx->global_ctx->capabilities.limits.maxComputeWorkgroupsPerDimension, wg_x, wg_y);1168 1169 return ggml_backend_webgpu_build(ctx, pipeline, params, entries, wg_x, wg_y);1170}1171 1172static webgpu_encoded_op ggml_webgpu_ssm_conv(webgpu_context & ctx,1173 ggml_tensor * src0,1174 ggml_tensor * src1,1175 ggml_tensor * dst) {1176 ggml_webgpu_shader_lib_context shader_lib_ctx = {};1177 shader_lib_ctx.src0 = src0;1178 shader_lib_ctx.src1 = src1;1179 shader_lib_ctx.dst = dst;1180 shader_lib_ctx.max_wg_size = ctx->global_ctx->capabilities.limits.maxComputeInvocationsPerWorkgroup;1181 1182 webgpu_pipeline pipeline = ctx->shader_lib->get_ssm_conv_pipeline(shader_lib_ctx);1183 auto * decisions = static_cast<ggml_webgpu_ssm_conv_shader_decisions *>(pipeline.context.get());1184 1185 const uint32_t token_tiles = CEIL_DIV((uint32_t) dst->ne[1], decisions->tokens_per_wg);1186 1187 std::vector<uint32_t> params = {1188 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src0) / ggml_type_size(src0->type)),1189 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, src1) / ggml_type_size(src1->type)),1190 (uint32_t) (ggml_webgpu_tensor_misalignment(ctx, dst) / ggml_type_size(dst->type)),1191 1192 (uint32_t) (src0->nb[1] / ggml_type_size(src0->type)),1193 (uint32_t) (src0->nb[2] / ggml_type_size(src0->type)),1194 (uint32_t) (src1->nb[1] / ggml_type_size(src1->type)),1195 1196 (uint32_t) (dst->nb[0] / ggml_type_size(dst->type)),1197 (uint32_t) (dst->nb[1] / ggml_type_size(dst->type)),1198 (uint32_t) (dst->nb[2] / ggml_type_size(dst->type)),1199 1200 (uint32_t) src1->ne[0],