Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#include "ggml-et.h"2 3#include "ggml-backend-impl.h"4#include "ggml-backend.h"5#include "ggml-et-common.h"6#include "ggml-et-kernels.h"7#include "ggml-et-memops.h"8#include "ggml-et-ops.h"9#include "ggml-impl.h"10#include "ggml.h"11 12#include <stdarg.h>13 14#include <cstdarg>15#include <cstdio>16#include <cstdlib>17#include <cstring>18#include <vector>19 20#if __has_include(<filesystem>)21# include <filesystem>22namespace fs = std::filesystem;23#elif __has_include(<experimental/filesystem>)24# include <experimental/filesystem>25namespace fs = std::experimental::filesystem;26#else27# error "cannot include the filesystem library"28#endif29 30/*31 * ggml_et_dump_tensor_metadata32 * @brief prints the metadata of a single tensorf33 */34static void ggml_et_dump_tensor_metadata(const ggml_tensor * ggtensor, size_t indent_level, const char * title) {35 char * spaces = (char *) alloca(indent_level + 1);36 memset(spaces, ' ', indent_level);37 spaces[indent_level] = '\0';38 fprintf(stderr,39 "%s%s: %s\n"40 "%s type: %s\n"41 "%s ne: %lld %lld %lld %lld\n"42 "%s nb: %zu %zu %zu %zu\n"43 "%s op: %s\n"44 "%s data: %p\n"45 "%s src0: %p\n",46 spaces, title, ggtensor->name, spaces, ggml_type_name(ggtensor->type), spaces, (long long) ggtensor->ne[0],47 (long long) ggtensor->ne[1], (long long) ggtensor->ne[2], (long long) ggtensor->ne[3], spaces,48 ggtensor->nb[0], ggtensor->nb[1], ggtensor->nb[2], ggtensor->nb[3], spaces, ggml_op_name(ggtensor->op),49 spaces, ggtensor->data, spaces, (void *) ggtensor->src[0]);50}51 52/*53 * ggml_et_dump_operator_metadata54 * @brief prints the metadata of a single tensor (or operator) including it's input and views55 */56static void ggml_et_dump_operator_metadata(const ggml_tensor * ggtensor) {57 GGML_ASSERT(ggtensor != NULL);58 ggml_et_dump_tensor_metadata(ggtensor, 0, "GGML tensor");59 for (int i = 0; i < GGML_MAX_SRC && ggtensor->src[i]; i++) {60 char arr[16];61 int n = snprintf(arr, sizeof(arr), "src[%i]->name", i);62 GGML_ASSERT((unsigned) n < sizeof(arr) && "printed too much data to stack buffer");63 ggml_et_dump_tensor_metadata(ggtensor->src[i], 2, arr);64 }65 if (ggtensor->view_src) {66 ggml_et_dump_tensor_metadata(ggtensor, 2, "view_src");67 }68}69 70static struct ggml_et_driver {71 std::shared_ptr<dev::IDeviceLayer> device_layer;72 std::shared_ptr<rt::IRuntime> runtime;73 std::unique_ptr<std::ofstream> profile_stream;74 std::unique_ptr<std::ofstream> kernel_id_stream;75 std::vector<std::pair<std::string, rt::KernelId>> kernel_map;76 bool profiling_enabled = false;77} _drv;78 79// Check at runtime environment variables for paths likely holding ET toolchain with sysemu elf files80static std::string ggml_et_get_default_et_path() {81 // List of environment variables to check in order of preference82 const char * const env_vars[] = { "ET_TOOLCHAIN", "TOOLCHAIN_ROOT" };83 84 for (const char * var : env_vars) {85 if (const char * et_path = std::getenv(var)) {86 if (et_path && *et_path != '\0') {87 return fs::path(et_path).string();88 }89 }90 }91 92 // Otherwise assume default93 return fs::path("/opt/et").string();94}95 96// config when using sysemu instead of PCIe hardware device97// adapted from `ainekko/et-platform/esperanto-tools-libs/tools/src/bench.cpp`98static inline auto ggml_et_get_default_sysemu_options() {99 constexpr uint64_t kSysEmuMaxCycles = std::numeric_limits<uint64_t>::max();100 constexpr uint64_t kSysEmuMinionShiresMask = 0x1FFFFFFFFu;101 const std::string et_path = ggml_et_get_default_et_path() + "/";102 103 emu::SysEmuOptions sysEmuOptions;104 105 // Construct all paths106 sysEmuOptions.bootromTrampolineToBL2ElfPath =107 et_path + "lib/esperanto-fw/BootromTrampolineToBL2/BootromTrampolineToBL2.elf";108 sysEmuOptions.spBL2ElfPath =109 et_path + "lib/esperanto-fw/ServiceProcessorBL2/fast-boot/ServiceProcessorBL2_fast-boot.elf";110 sysEmuOptions.machineMinionElfPath = et_path + "lib/esperanto-fw/MachineMinion/MachineMinion.elf";111 sysEmuOptions.masterMinionElfPath = et_path + "lib/esperanto-fw/MasterMinion/MasterMinion.elf";112 sysEmuOptions.workerMinionElfPath = et_path + "lib/esperanto-fw/WorkerMinion/WorkerMinion.elf";113 sysEmuOptions.executablePath = et_path + "bin/sys_emu";114 115 // Check that each path has a valid existing non-zero file otherwise emulator just silently hangs116 const std::vector<std::string> required_files = {117 sysEmuOptions.bootromTrampolineToBL2ElfPath, sysEmuOptions.spBL2ElfPath,118 sysEmuOptions.machineMinionElfPath, sysEmuOptions.masterMinionElfPath,119 sysEmuOptions.workerMinionElfPath, sysEmuOptions.executablePath,120 };121 122 for (const auto & file : required_files) {123 if (!fs::exists(file) || fs::file_size(file) == 0) {124 // Check that each path has a valid existing non-zero file otherwise emulator just silently hangs125 GGML_LOG_ERROR("ET: Unable to find required sysemu file: %s", file.c_str());126 GGML_LOG_ERROR("ET: Confirm et-platform is correctly installed at configured path.");127 abort();128 }129 }130 131 sysEmuOptions.runDir = (fs::current_path().string() + "/");132 sysEmuOptions.maxCycles = kSysEmuMaxCycles;133 sysEmuOptions.minionShiresMask = kSysEmuMinionShiresMask;134 sysEmuOptions.puUart0Path = sysEmuOptions.runDir + "pu_uart0_tx.log";135 sysEmuOptions.puUart1Path = sysEmuOptions.runDir + "pu_uart1_tx.log";136 sysEmuOptions.spUart0Path = sysEmuOptions.runDir + "spio_uart0_tx.log";137 sysEmuOptions.spUart1Path = sysEmuOptions.runDir + "spio_uart1_tx.log";138 sysEmuOptions.startGdb = false;139 sysEmuOptions.memcheck = false;140 141 return sysEmuOptions;142}143 144// Forward declaration145static void ggml_et_driver_cleanup();146 147static bool ggml_et_driver_init() {148 if (_drv.runtime != nullptr) {149 assert(_drv.device_layer != nullptr);150 } else {151 try {152#if defined GGML_ET_SYSEMU && GGML_ET_SYSEMU153 // For emulator device using sysEmuOptions provided by function above enabled compiling with `-DGGML_ET_SYSEMU=ON`154 _drv.device_layer = dev::IDeviceLayer::createSysEmuDeviceLayer(ggml_et_get_default_sysemu_options());155#else156 // For physical PCIe device157 _drv.device_layer = dev::IDeviceLayer::createPcieDeviceLayer();158#endif // GGML_ET_SYSEMU159 160 _drv.runtime = rt::IRuntime::create(_drv.device_layer);161 162 // Initialize profiler if requested via environment variable163 const char * profile_path = getenv("GGML_ET_PROFILE");164 if (profile_path) {165 std::string output_path = std::string(profile_path) + "/et_runtime_trace.json";166 std::string kernel_id_path = std::string(profile_path) + "/kernel_id.json";167 168 _drv.profile_stream = std::make_unique<std::ofstream>(output_path);169 _drv.kernel_id_stream = std::make_unique<std::ofstream>(kernel_id_path);170 if (!_drv.profile_stream->is_open()) {171 GGML_LOG_ERROR("ET: Failed to open profiling output file: %s", output_path.c_str());172 abort();173 }174 if (!_drv.kernel_id_stream->is_open()) {175 GGML_LOG_ERROR("ET: Failed to open profiling kernel map: %s", kernel_id_path.c_str());176 abort();177 }178 179 auto * profiler = _drv.runtime->getProfiler();180 profiler->start(*_drv.profile_stream, rt::IProfiler::OutputType::Json);181 _drv.profiling_enabled = true;182 GGML_LOG_INFO("ET: Runtime profiler started (JSON format)");183 184 // Register cleanup at program exit185 std::atexit(ggml_et_driver_cleanup);186 }187 } catch (const std::exception & e) {188 GGML_LOG_ERROR("ggml_et: %s", e.what());189 if (_drv.device_layer != nullptr) {190 _drv.device_layer.reset();191 }192 if (_drv.runtime != nullptr) {193 _drv.runtime.reset();194 }195 return false;196 }197 }198 return true;199}200 201static std::shared_ptr<dev::IDeviceLayer> ggml_et_devicelayer() {202 return _drv.device_layer;203}204 205std::shared_ptr<rt::IRuntime> ggml_et_runtime() {206 return _drv.runtime;207}208 209static void ggml_et_driver_cleanup() {210 if (_drv.profiling_enabled && _drv.runtime) {211 GGML_LOG_INFO("ET: Stopping runtime profiler");212 auto * profiler = _drv.runtime->getProfiler();213 profiler->stop();214 _drv.profiling_enabled = false;215 216 if (_drv.profile_stream) {217 _drv.profile_stream->close();218 _drv.profile_stream.reset();219 }220 221 // Save kernel map222 if (_drv.kernel_id_stream && !_drv.kernel_map.empty()) {223 auto & os = *_drv.kernel_id_stream;224 // XXX: Manual JSON construction. Not pretty but removes dependency225 os << "{\n";226 for (size_t i = 0; i < _drv.kernel_map.size(); i++) {227 os << " \"" << _drv.kernel_map[i].first << "\": " << (int) _drv.kernel_map[i].second;228 if (i + 1 < _drv.kernel_map.size()) {229 os << ",";230 }231 os << "\n";232 }233 os << "}\n";234 _drv.kernel_id_stream->close();235 _drv.kernel_id_stream.reset();236 }237 }238}239 240static ggml_backend_dev_t ggml_backend_et_reg_get_device(ggml_backend_reg_t reg, size_t devidx);241 242static void ggml_backend_et_buffer_free_buffer(ggml_backend_buffer_t buffer) {243 ggml_backend_et_buffer_context * ctx = (ggml_backend_et_buffer_context *) buffer->context;244 if (ctx->data != nullptr) {245 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();246 if (runtime) {247 runtime->freeDevice(ctx->rtid, static_cast<std::byte *>(ctx->data));248 }249 }250 delete ctx;251}252 253static void * ggml_backend_et_buffer_get_base(ggml_backend_buffer_t buffer) {254 ggml_backend_et_buffer_context * ctx = (ggml_backend_et_buffer_context *) buffer->context;255 return ctx->data;256}257 258static ggml_status ggml_backend_et_buffer_init_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor) {259 // View tensors share buffer with their view_src, no additional initialization needed260 if (tensor->view_src != NULL) {261 return GGML_STATUS_SUCCESS;262 }263 264 const size_t original_size = ggml_nbytes(tensor);265 const size_t padded_size = ggml_backend_buft_get_alloc_size(buffer->buft, tensor);266 267 // Clear padding bytes to avoid NaN values268 // XXX: Martin - do we need this?269 if (padded_size > original_size) {270 const size_t padding_size = padded_size - original_size;271 272 // Get device context to access memops kernel273 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buffer->buft->device->context;274 if (!dev_ctx) {275 GGML_LOG_ERROR("ET: Failed to get device context for padding clear");276 return GGML_STATUS_FAILED;277 }278 279 // Use device-side memset kernel for efficient padding clear280 std::byte * padding_ptr = static_cast<std::byte *>(tensor->data) + original_size;281 if (!ggml_et_memset(dev_ctx, padding_ptr, 0, padding_size)) {282 GGML_LOG_ERROR("ET: Failed to clear padding using memset kernel for tensor %s", tensor->name);283 return GGML_STATUS_FAILED;284 }285 }286 287 return GGML_STATUS_SUCCESS;288}289 290static void ggml_backend_et_buffer_set_tensor(ggml_backend_buffer_t buffer,291 ggml_tensor * tensor,292 const void * data,293 size_t offset,294 size_t size) {295 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();296 if (!runtime) {297 return;298 }299 300 // Create short-lived stream for this transfer301 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buffer->buft->device->context;302 rt::StreamId stream = dev_ctx->default_stream;303 304 std::byte * dst_ptr = static_cast<std::byte *>(tensor->data) + offset;305 const std::byte * src_ptr = static_cast<const std::byte *>(data);306 307 rt::EventId event = runtime->memcpyHostToDevice(stream, src_ptr, dst_ptr, size, true /*barrier*/);308 309 runtime->waitForEvent(event);310}311 312static void ggml_backend_et_buffer_get_tensor(ggml_backend_buffer_t buffer,313 const ggml_tensor * tensor,314 void * data,315 size_t offset,316 size_t size) {317 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();318 if (!runtime) {319 return;320 }321 322 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buffer->buft->device->context;323 rt::StreamId stream = dev_ctx->default_stream;324 325 const std::byte * src_ptr = static_cast<const std::byte *>(tensor->data) + offset;326 std::byte * dst_ptr = static_cast<std::byte *>(data);327 328 rt::EventId event = runtime->memcpyDeviceToHost(stream, src_ptr, dst_ptr, size, true /*barrier*/);329 330 runtime->waitForEvent(event);331}332 333static bool ggml_backend_et_buffer_cpy_tensor(ggml_backend_buffer_t buffer,334 const ggml_tensor * src,335 ggml_tensor * dst) {336 GGML_UNUSED(buffer);337 GGML_UNUSED(src);338 GGML_UNUSED(dst);339 return false;340}341 342static void ggml_backend_et_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) {343 ggml_backend_et_buffer_context * ctx = (ggml_backend_et_buffer_context *) buffer->context;344 345 if (ctx->size == 0 || ctx->data == nullptr) {346 return;347 }348 349 // Get device context to access memops kernel350 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buffer->buft->device->context;351 if (!dev_ctx) {352 GGML_LOG_ERROR("ET: Failed to get device context for buffer clear");353 return;354 }355 356 // Use device-side memset kernel for efficient clearing357 if (!ggml_et_memset(dev_ctx, ctx->data, value, ctx->size)) {358 GGML_LOG_ERROR("ET: buffer_clear failed using memset kernel");359 return;360 }361 362 GGML_LOG_DEBUG("ET: Buffer cleared successfully using memops kernel");363}364 365static const struct ggml_backend_buffer_i ggml_backend_et_buffer_i = {366 /* .free_buffer = */ ggml_backend_et_buffer_free_buffer,367 /* .get_base = */ ggml_backend_et_buffer_get_base,368 /* .init_tensor = */ ggml_backend_et_buffer_init_tensor,369 /* .memset_tensor = */ NULL,370 /* .set_tensor = */ ggml_backend_et_buffer_set_tensor,371 /* .get_tensor = */ ggml_backend_et_buffer_get_tensor,372 /* .set_tensor_2d = */ NULL,373 /* .get_tensor_2d = */ NULL,374 /* .cpy_tensor = */ ggml_backend_et_buffer_cpy_tensor,375 /* .clear = */ ggml_backend_et_buffer_clear,376 /* .reset = */ NULL,377};378 379static const char * ggml_backend_et_buffer_type_get_name(ggml_backend_buffer_type_t buft) {380 GGML_UNUSED(buft);381 return GGML_ET_NAME;382}383 384static ggml_backend_buffer_t ggml_backend_et_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) {385 ggml_backend_et_buffer_type_context * btctx = (ggml_backend_et_buffer_type_context *) buft->context;386 387 ggml_backend_et_buffer_context * ctx = new ggml_backend_et_buffer_context;388 ctx->devidx = btctx->devidx;389 ctx->size = size;390 391 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();392 if (!runtime) {393 delete ctx;394 return nullptr;395 }396 397 std::vector<rt::DeviceId> rtids = runtime->getDevices();398 if (static_cast<size_t>(btctx->devidx) >= rtids.size()) {399 delete ctx;400 return nullptr;401 }402 ctx->rtid = rtids[btctx->devidx];403 404 ctx->data = runtime->mallocDevice(ctx->rtid, size);405 if (ctx->data == nullptr) {406 delete ctx;407 return nullptr;408 }409 410 return ggml_backend_buffer_init(buft, ggml_backend_et_buffer_i, ctx, size);411}412 413static size_t ggml_backend_et_buffer_type_get_alignment(ggml_backend_buffer_type_t buft) {414 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();415 if (!runtime || !buft->device) {416 return GGML_MEM_ALIGN;417 }418 419 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buft->device->context;420 rt::DeviceProperties prop = runtime->getDeviceProperties(dev_ctx->rtid);421 return prop.cacheLineSize_;422}423 424static size_t ggml_backend_et_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) {425 if (buft->device) {426 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buft->device->context;427 return dev_ctx->total_mem;428 }429 return SIZE_MAX;430}431 432static size_t ggml_backend_et_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const ggml_tensor * tensor) {433 GGML_UNUSED(buft);434 return ggml_nbytes_pad(tensor);435}436 437static bool ggml_backend_et_buffer_type_is_host(ggml_backend_buffer_type_t buft) {438 GGML_UNUSED(buft);439 return false;440}441 442static const struct ggml_backend_buffer_type_i ggml_backend_et_buffer_type_i = {443 /* .get_name = */ ggml_backend_et_buffer_type_get_name,444 /* .alloc_buffer = */ ggml_backend_et_buffer_type_alloc_buffer,445 /* .get_alignment = */ ggml_backend_et_buffer_type_get_alignment,446 /* .get_max_size = */ ggml_backend_et_buffer_type_get_max_size,447 /* .get_alloc_size = */ ggml_backend_et_buffer_type_get_alloc_size,448 /* .is_host = */ ggml_backend_et_buffer_type_is_host,449};450 451static const char * ggml_backend_et_get_name(ggml_backend_t backend) {452 GGML_UNUSED(backend);453 return GGML_ET_NAME;454}455 456static void ggml_backend_et_free(ggml_backend_t backend) {457 ggml_backend_et_context * et_ctx = (ggml_backend_et_context *) backend->context;458 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();459 460 // Clean up kernels on this device before freeing backend461 ggml_backend_dev_t dev = ggml_backend_et_reg_get_device(ggml_backend_et_reg(), et_ctx->devidx);462 if (dev && dev->context) {463 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) dev->context;464 465 if (_drv.profiling_enabled) {466 auto kernels = ggml_et_get_loaded_kernels(dev_ctx);467 _drv.kernel_map.insert(_drv.kernel_map.end(), kernels.begin(), kernels.end());468 }469 470 ggml_et_unload_all_kernels(dev_ctx);471 472 if (runtime) {473 if (dev_ctx->trace_buffer) {474 runtime->freeDevice(dev_ctx->rtid, dev_ctx->trace_buffer);475 dev_ctx->trace_buffer = nullptr;476 }477 // Drain any in-flight uberkernel launches before freeing the478 // device buffers they read from.479 runtime->waitForStream(dev_ctx->default_stream);480 for (auto & slot : dev_ctx->uberkernel.slots) {481 if (slot.device_insts) {482 runtime->freeDevice(dev_ctx->rtid, slot.device_insts);483 slot.device_insts = nullptr;484 }485 if (slot.device_params) {486 runtime->freeDevice(dev_ctx->rtid, slot.device_params);487 slot.device_params = nullptr;488 }489 slot.has_pending = false;490 }491 }492 }493 494 delete et_ctx;495 delete backend;496}497 498static ggml_backend_buffer_type_t ggml_backend_et_get_default_buffer_type(ggml_backend_t backend) {499 ggml_backend_et_context * et_ctx = (ggml_backend_et_context *) backend->context;500 501 return ggml_backend_et_buffer_type(et_ctx->devidx);502}503 504static void ggml_backend_et_set_tensor_async(ggml_backend_t backend,505 ggml_tensor * tensor,506 const void * data,507 size_t offset,508 size_t size) {509 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();510 if (!runtime) {511 return;512 }513 514 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) backend->device->context;515 rt::StreamId stream = dev_ctx->default_stream;516 517 std::byte * dst_ptr = static_cast<std::byte *>(tensor->data) + offset;518 const std::byte * src_ptr = static_cast<const std::byte *>(data);519 520 runtime->memcpyHostToDevice(stream, src_ptr, dst_ptr, size, true /*barrier*/);521}522 523static void ggml_backend_et_get_tensor_async(ggml_backend_t backend,524 const ggml_tensor * tensor,525 void * data,526 size_t offset,527 size_t size) {528 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();529 if (!runtime) {530 return;531 }532 533 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) backend->device->context;534 rt::StreamId stream = dev_ctx->default_stream;535 536 const std::byte * src_ptr = static_cast<const std::byte *>(tensor->data) + offset;537 std::byte * dst_ptr = static_cast<std::byte *>(data);538 539 runtime->memcpyDeviceToHost(stream, src_ptr, dst_ptr, size, true /*barrier*/);540}541 542static bool ggml_backend_et_cpy_tensor_async(ggml_backend_t backend_src,543 ggml_backend_t backend_dst,544 const ggml_tensor * src,545 ggml_tensor * dst) {546 GGML_UNUSED(backend_src);547 GGML_UNUSED(backend_dst);548 GGML_UNUSED(src);549 GGML_UNUSED(dst);550 return false;551}552 553static void ggml_backend_et_synchronize(ggml_backend_t backend) {554 std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();555 if (!runtime) {556 return;557 }558 559 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) backend->device->context;560 runtime->waitForStream(dev_ctx->default_stream);561 562 auto errors = runtime->retrieveStreamErrors(dev_ctx->default_stream);563 if (errors.empty()) {564 return;565 }566 for (const auto & err : errors) {567 GGML_LOG_ERROR("ET: stream error detected at synchronization point. Code: %d,Type: %d\n", (int) err.errorCode_,568 (int) err.errorContext_.value()[0].type_);569 }570 abort();571}572 573static bool ggml_et_can_fuse(const ggml_cgraph * cgraph, int node_idx, std::initializer_list<ggml_op> ops) {574 if (!ggml_can_fuse(cgraph, node_idx, ops)) {575 return false;576 }577 578 if (ops.size() == 2 && ops.begin()[0] == GGML_OP_MUL_MAT && ops.begin()[1] == GGML_OP_ADD) {579 const ggml_tensor * mm = cgraph->nodes[node_idx];580 const ggml_tensor * add = cgraph->nodes[node_idx + 1];581 582 // Only Q8_0 weights x F32 activations -> F32 (the kernel that has583 // the bias path). Other MM variants must wait for their own kernel584 // bias support.585 if (mm->type != GGML_TYPE_F32 || mm->src[0]->type != GGML_TYPE_Q8_0 || mm->src[1]->type != GGML_TYPE_F32) {586 return false;587 }588 589 // ADD must be F32 and one of its operands must be the MM output.590 if (add->type != GGML_TYPE_F32) {591 return false;592 }593 if (add->src[0] != mm && add->src[1] != mm) {594 return false;595 }596 597 const ggml_tensor * bias = (add->src[0] == mm) ? add->src[1] : add->src[0];598 599 if (bias->type != GGML_TYPE_F32) {600 return false;601 }602 603 // No broadcasting: bias shape must equal MM output shape.604 for (int i = 0; i < GGML_MAX_DIMS; ++i) {605 if (bias->ne[i] != mm->ne[i]) {606 return false;607 }608 }609 610 // Bias and dst must be contiguous and have identical strides - the611 // kernel uses dst-style offset arithmetic against bias's nb[].612 if (!ggml_is_contiguous(bias) || !ggml_is_contiguous(mm)) {613 return false;614 }615 for (int i = 0; i < GGML_MAX_DIMS; ++i) {616 if ((int64_t) bias->nb[i] != (int64_t) add->nb[i]) {617 return false;618 }619 }620 }621 622 if (ops.size() == 2 && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) {623 const ggml_tensor * rms_norm = cgraph->nodes[node_idx];624 const ggml_tensor * mul = cgraph->nodes[node_idx + 1];625 626 // ET only supports F32627 if (rms_norm->src[0]->type != GGML_TYPE_F32 || mul->type != GGML_TYPE_F32) {628 return false;629 }630 631 // Identify the weights tensor (the MUL operand that isn't rms_norm output)632 const ggml_tensor * weights = (mul->src[0] == rms_norm) ? mul->src[1] : mul->src[0];633 634 if (weights->type != GGML_TYPE_F32) {635 return false;636 }637 638 // Both inputs must be contiguous (ET hardware requirement)639 if (!ggml_is_contiguous(rms_norm->src[0]) || !ggml_is_contiguous_rows(weights)) {640 return false;641 }642 643 // ET requires cache-aligned rows (ne[0] % 16 == 0)644 if (rms_norm->src[0]->ne[0] % 16 != 0 || weights->ne[0] % 16 != 0) {645 return false;646 }647 648 // Fused kernel doesn't handle dim-0 broadcasting649 if (weights->ne[0] != rms_norm->src[0]->ne[0]) {650 return false;651 }652 }653 654 return true;655}656 657static ggml_status ggml_backend_et_graph_compute(ggml_backend_t backend, ggml_cgraph * cgraph) {658 ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) backend->device->context;659 ggml_et_uberkernel_begin_graph(&dev_ctx->uberkernel);660 661 for (int i = 0; i < cgraph->n_nodes; i++) {662 ggml_tensor * node = cgraph->nodes[i];663 664 if (node->op == GGML_OP_NONE || node->op == GGML_OP_VIEW || node->op == GGML_OP_RESHAPE ||665 node->op == GGML_OP_PERMUTE || node->op == GGML_OP_TRANSPOSE) {666 continue;667 }668 669 // --- Fusion checks (before regular dispatch) ---670 if (ggml_et_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL })) {671 ggml_et_op_rms_norm_mul(dev_ctx, node, cgraph->nodes[i + 1]);672 i++; // skip the MUL node673 continue;674 }675 if (ggml_et_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD })) {676 ggml_et_op_mul_mat(dev_ctx, node, cgraph->nodes[i + 1]);677 i++; // skip the ADD node678 continue;679 }680 681 switch (node->op) {682 case GGML_OP_SQR:683 ggml_et_op_sqr(dev_ctx, node);684 break;685 686 case GGML_OP_UNARY:687 ggml_et_op_unary(dev_ctx, node);688 break;689 690 case GGML_OP_SUM_ROWS:691 ggml_et_op_sum_rows(dev_ctx, node);692 break;693 694 case GGML_OP_MEAN:695 ggml_et_op_mean(dev_ctx, node);696 break;697 698 case GGML_OP_CLAMP:699 ggml_et_op_clamp(dev_ctx, node);700 break;701 702 case GGML_OP_MUL:703 ggml_et_op_mul(dev_ctx, node);704 break;705 706 case GGML_OP_ADD:707 ggml_et_op_add(dev_ctx, node);708 break;709 710 case GGML_OP_SUB:711 ggml_et_op_sub(dev_ctx, node);712 break;713 714 case GGML_OP_CUMSUM:715 ggml_et_op_cumsum(dev_ctx, node);716 break;717 718 case GGML_OP_MUL_MAT:719 ggml_et_op_mul_mat(dev_ctx, node);720 break;721 722 case GGML_OP_MUL_MAT_ID:723 ggml_et_op_mul_mat_id(dev_ctx, node);724 break;725 726 case GGML_OP_ROPE:727 ggml_et_op_rope(dev_ctx, node);728 break;729 730 case GGML_OP_RMS_NORM:731 ggml_et_op_rms_norm(dev_ctx, node);732 break;733 734 case GGML_OP_NORM:735 ggml_et_op_norm(dev_ctx, node);736 break;737 738 case GGML_OP_L2_NORM:739 ggml_et_op_l2_norm(dev_ctx, node);740 break;741 742 case GGML_OP_GROUP_NORM:743 ggml_et_op_group_norm(dev_ctx, node);744 break;745 746 case GGML_OP_SCALE:747 ggml_et_op_scale(dev_ctx, node);748 break;749 750 case GGML_OP_GLU:751 ggml_et_op_glu(dev_ctx, node);752 break;753 754 case GGML_OP_SOFT_MAX:755 ggml_et_op_softmax(dev_ctx, node);756 break;757 758 case GGML_OP_IM2COL:759 ggml_et_op_im2col(dev_ctx, node);760 break;761 762 case GGML_OP_CONV_2D:763 ggml_et_op_conv_2d(dev_ctx, node);764 break;765 766 case GGML_OP_FLASH_ATTN_EXT:767 ggml_et_op_flash_attn_ext(dev_ctx, node);768 break;769 770 case GGML_OP_GET_ROWS:771 ggml_et_op_get_rows(dev_ctx, node);772 break;773 774 case GGML_OP_CONT:775 ggml_et_op_cont(dev_ctx, node);776 break;777 778 case GGML_OP_CPY:779 ggml_et_op_cpy(dev_ctx, node);780 break;781 782 case GGML_OP_CONCAT:783 ggml_et_op_concat(dev_ctx, node);784 break;785 786 case GGML_OP_REPEAT:787 ggml_et_op_repeat(dev_ctx, node);788 break;789 790 case GGML_OP_SSM_CONV:791 ggml_et_op_ssm_conv(dev_ctx, node);792 break;793 794 case GGML_OP_SSM_SCAN:795 ggml_et_op_ssm_scan(dev_ctx, node);796 break;797 798 case GGML_OP_PAD:799 ggml_et_op_pad(dev_ctx, node);800 break;801 802 case GGML_OP_SET_ROWS:803 ggml_et_op_set_rows(dev_ctx, node);804 break;805 806 case GGML_OP_FILL:807 ggml_et_op_fill(dev_ctx, node);808 break;809 810 case GGML_OP_DIAG:811 ggml_et_op_diag(dev_ctx, node);812 break;813 814 case GGML_OP_TRI:815 ggml_et_op_tri(dev_ctx, node);816 break;817 818 case GGML_OP_SOLVE_TRI:819 ggml_et_op_solve_tri(dev_ctx, node);820 break;821 822 case GGML_OP_SET:823 ggml_et_op_set(dev_ctx, node);824 break;825 826 case GGML_OP_RWKV_WKV6:827 ggml_et_op_rwkv_wkv6(dev_ctx, node);828 break;829 830 case GGML_OP_RWKV_WKV7:831 ggml_et_op_rwkv_wkv7(dev_ctx, node);832 break;833 834 case GGML_OP_GATED_DELTA_NET:835 ggml_et_op_gated_delta_net(dev_ctx, node);836 break;837 838 default:839 ggml_et_uberkernel_abort_graph(&dev_ctx->uberkernel);840 GGML_LOG_ERROR("ET: Unsupported operation in graph: %s", ggml_op_name(node->op));841 return GGML_STATUS_FAILED;842 }843 844 if (ggml_et_uberkernel_failed(&dev_ctx->uberkernel)) {845 ggml_et_uberkernel_abort_graph(&dev_ctx->uberkernel);846 return GGML_STATUS_FAILED;847 }848 }849 850 if (!ggml_et_uberkernel_end_graph(dev_ctx)) {851 ggml_et_uberkernel_abort_graph(&dev_ctx->uberkernel);852 return GGML_STATUS_FAILED;853 }854 855 return GGML_STATUS_SUCCESS;856}857 858// Check that elements within each row are contiguous (nb[0] == type_size).859// Higher-dim strides can be arbitrary - kernels navigate them via byte offsets.860static bool et_ggml_is_row_contiguous(const ggml_tensor * t) {861 return t->nb[0] == ggml_type_size(t->type);862}863 864static bool ggml_backend_et_device_supports_op(ggml_backend_dev_t dev, const ggml_tensor * op) {865 GGML_UNUSED(dev);866 867 bool supported = false;868 switch (op->op) {869 case GGML_OP_CUMSUM:870 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&871 op->src[0]->nb[0] == sizeof(float) && ggml_is_contiguous(op);872 break;873 case GGML_OP_SQR:874 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&875 op->ne[0] % 16 == 0 && ggml_is_contiguous(op) && ggml_is_contiguous(op->src[0]);876 break;877 case GGML_OP_SUM_ROWS:878 // dst has ne[0]=1, src0 row length must be cache-aligned879 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&880 op->src[0]->ne[0] % 16 == 0 && ggml_is_contiguous(op->src[0]);881 break;882 case GGML_OP_MEAN:883 // Kernel handles arbitrary ne00 (per-row alignment guard with884 // scalar tail), so no row-length divisibility constraint here.885 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&886 ggml_is_contiguous(op->src[0]);887 break;888 case GGML_OP_CLAMP:889 // Element-wise; kernel distributes by cache lines and handles a890 // scalar tail, so any contiguous F32 size is fine - including the891 // 1x1x1x1 scalar case.892 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&893 ggml_is_contiguous(op) && ggml_is_contiguous(op->src[0]);894 break;895 case GGML_OP_UNARY:896 // Only require dim-0 contiguity (nb[0] == sizeof(float)). Higher897 // dims may be arbitrarily strided views; the kernel walks per-row898 // using all four nb[] values. See unary_f32.c entry_point.899 if (op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&900 ggml_nelements(op) % 16 == 0 && op->nb[0] == sizeof(float) && op->src[0]->nb[0] == sizeof(float)) {901 switch (ggml_get_unary_op(op)) {902 case GGML_UNARY_OP_ABS:903 case GGML_UNARY_OP_SGN:904 case GGML_UNARY_OP_NEG:905 case GGML_UNARY_OP_STEP:906 case GGML_UNARY_OP_TANH:907 case GGML_UNARY_OP_ELU:908 case GGML_UNARY_OP_RELU:909 case GGML_UNARY_OP_SIGMOID:910 case GGML_UNARY_OP_GELU:911 case GGML_UNARY_OP_GELU_QUICK:912 case GGML_UNARY_OP_SILU:913 case GGML_UNARY_OP_HARDSWISH:914 case GGML_UNARY_OP_HARDSIGMOID:915 case GGML_UNARY_OP_EXP:916 case GGML_UNARY_OP_EXPM1:917 case GGML_UNARY_OP_SOFTPLUS:918 case GGML_UNARY_OP_GELU_ERF:919 case GGML_UNARY_OP_FLOOR:920 case GGML_UNARY_OP_CEIL:921 case GGML_UNARY_OP_ROUND:922 case GGML_UNARY_OP_TRUNC:923 supported = true;924 break;925 default:926 break;927 }928 }929 break;930 case GGML_OP_MUL:931 case GGML_OP_ADD:932 case GGML_OP_SUB:933 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 && op->src[1] &&934 op->src[1]->type == GGML_TYPE_F32 && op->nb[0] == sizeof(float) &&935 op->src[0]->nb[0] == sizeof(float) &&936 (op->src[1]->nb[0] == sizeof(float) || op->src[1]->ne[0] == 1) &&937 op->nb[1] == op->ne[0] * sizeof(float);938 break;939 case GGML_OP_MUL_MAT:940 // Support Q8_0 x F32 -> F32, F16 x F32 -> F32, F16 x F16 -> F32, and F32 x F32 -> F32 matrix multiplication941 // Stride requirements: first dimension must be contiguous for all tensors942 if (op->type == GGML_TYPE_F32 &&943 ((op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32) ||944 (op->src[0]->type == GGML_TYPE_F16 && op->src[1]->type == GGML_TYPE_F16)) &&945 op->ne[0] % 16 == 0 && // dst row length for tensor-store path946 op->src[0]->ne[1] % 16 == 0 && // m947 op->src[0]->ne[0] % 16 == 0 && // k948 ggml_is_contiguous(op->src[0]) && ggml_is_contiguous(op->src[1])) {949 // Special path for the FP32 TensorFMA kernel950 // Limitation - generic kernels can tolerate non-cache-aligned dst rows951 // because they publish each output element atomically. The matrix952 // engine path still uses tiled tensor stores, so keep dst rows aligned.953 // The m edge is difficult to do because of the 4 conseqtive load hardware limitation954 // And the k edge is impossible because that is encoded as `stride & 0xFFFFFFFFFFC0ULL` which becomes 0 for stride 16 (4x FP32) :(955 // FIXME: Right now this overwrites the mul_mat_f32 kernel - whatever. Fix later. Demo code956 supported = true;957 } else if (op->type == GGML_TYPE_F32 && op->src[0] &&958 (op->src[0]->type == GGML_TYPE_F16 || op->src[0]->type == GGML_TYPE_F32) && op->src[1] &&959 (op->src[1]->type == GGML_TYPE_F16 || op->src[1]->type == GGML_TYPE_F32)) {960 // Check first dimension contiguity requirements961 bool src0_first_dim_contiguous = (op->src[0]->nb[0] == ggml_type_size(op->src[0]->type));962 bool src1_first_dim_contiguous = (op->src[1]->nb[0] == ggml_type_size(op->src[1]->type));963 bool dst_first_dim_contiguous = (op->nb[0] == sizeof(float));964 965 // Check destination stride ordering (only for dimensions with ne > 1)966 bool dst_properly_ordered = true;967 for (int d = 0; d < 3; d++) {968 if (op->ne[d] > 1 && op->ne[d + 1] > 1 && op->nb[d] > op->nb[d + 1]) {969 dst_properly_ordered = false;970 }971 }972 973 supported = src0_first_dim_contiguous && src1_first_dim_contiguous && dst_first_dim_contiguous &&974 dst_properly_ordered;975 } else if (op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_Q8_0 && op->src[1] &&976 op->src[1]->type == GGML_TYPE_F32) {977 // Keep the existing quantized path constraints separate from the978 // relaxed non-quant generic fallback.979 bool src0_first_dim_contiguous = (op->src[0]->nb[0] == ggml_type_size(op->src[0]->type));980 bool src1_first_dim_contiguous = (op->src[1]->nb[0] == ggml_type_size(op->src[1]->type));981 bool dst_first_dim_contiguous = (op->nb[0] == sizeof(float));982 983 bool dst_properly_ordered = true;984 for (int d = 0; d < 3; d++) {985 if (op->ne[d] > 1 && op->ne[d + 1] > 1 && op->nb[d] > op->nb[d + 1]) {986 dst_properly_ordered = false;987 }988 }989 990 supported = src0_first_dim_contiguous && src1_first_dim_contiguous && dst_first_dim_contiguous &&991 dst_properly_ordered;992 993 } else if (op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_Q4_0 && op->src[1] &&994 op->src[1]->type == GGML_TYPE_F32) {995 // Keep the existing quantized path constraints separate from the996 // relaxed non-quant generic fallback.997 bool src0_first_dim_contiguous = (op->src[0]->nb[0] == ggml_type_size(op->src[0]->type));998 bool src1_first_dim_contiguous = (op->src[1]->nb[0] == ggml_type_size(op->src[1]->type));999 bool dst_first_dim_contiguous = (op->nb[0] == sizeof(float));1000 1001 bool dst_properly_ordered = true;1002 for (int d = 0; d < 3; d++) {1003 if (op->ne[d] > 1 && op->ne[d + 1] > 1 && op->nb[d] > op->nb[d + 1]) {1004 dst_properly_ordered = false;1005 }1006 }1007 1008 supported = src0_first_dim_contiguous && src1_first_dim_contiguous && dst_first_dim_contiguous &&1009 dst_properly_ordered;1010 } else {1011 supported = false;1012 }1013 break;1014 case GGML_OP_MUL_MAT_ID:1015 // Support MUL_MAT_ID for Mixture of Experts: (Q8_0/Q4_0/F16/F32) x F32 -> F32 with I32 expert indices1016 // src0 (as): [K, M, n_expert] - expert weight matrices (can be quantized)1017 // src1 (b): [K, n_expert_used, batch] - activations (F32)1018 // src2 (ids): [n_expert_used, batch] - expert selection indices (I32)1019 // dst: [M, n_expert_used, batch, 1] - output (F32)1020 if (op->type == GGML_TYPE_F32 && op->src[0] &&1021 (op->src[0]->type == GGML_TYPE_Q8_0 || op->src[0]->type == GGML_TYPE_Q4_0 ||1022 op->src[0]->type == GGML_TYPE_F16 || op->src[0]->type == GGML_TYPE_F32) &&1023 op->src[1] && op->src[1]->type == GGML_TYPE_F32 && op->src[2] && op->src[2]->type == GGML_TYPE_I32) {1024 // Check first dimension contiguity requirements (matching CPU backend)1025 bool src0_first_dim_contiguous = (op->src[0]->nb[0] == ggml_type_size(op->src[0]->type));1026 bool src1_first_dim_contiguous = (op->src[1]->nb[0] == ggml_type_size(op->src[1]->type));1027 bool src2_first_dim_contiguous = (op->src[2]->nb[0] == ggml_type_size(op->src[2]->type));1028 bool dst_first_dim_contiguous = (op->nb[0] == sizeof(float));1029 1030 // Check destination stride ordering (only for dimensions with ne > 1)1031 bool dst_properly_ordered = true;1032 for (int d = 0; d < 3; d++) {1033 if (op->ne[d] > 1 && op->ne[d + 1] > 1 && op->nb[d] > op->nb[d + 1]) {1034 dst_properly_ordered = false;1035 }1036 }1037 1038 // Validate tensor dimension constraints from GGML definition1039 bool dims_valid = (op->src[0]->ne[3] == 1) && // as is 3d (one matrix per expert)1040 (op->src[1]->ne[3] == 1) && // b is 3d1041 (op->src[2]->ne[2] == 1 && op->src[2]->ne[3] == 1) && // ids is 2d1042 (op->src[2]->ne[1] == op->src[1]->ne[2]) && // must have expert list per b row1043 (op->src[0]->ne[0] == op->src[1]->ne[0]) && // K dimension must match1044 (op->src[2]->ne[0] % op->src[1]->ne[1] == 0); // can broadcast1045 1046 supported = src0_first_dim_contiguous && src1_first_dim_contiguous && src2_first_dim_contiguous &&1047 dst_first_dim_contiguous && dst_properly_ordered && dims_valid;1048 } else {1049 supported = false;1050 }1051 break;1052 case GGML_OP_ROPE:1053 // Support F32 x I32 -> F32 RoPE for the modes implemented by rope_f32.1054 if (op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 && op->src[1] &&1055 op->src[1]->type == GGML_TYPE_I32 && ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0])) {1056 const int mode = ggml_get_op_params_i32(op, 2);1057 const int ndims = ggml_get_op_params_i32(op, 1);1058 const bool is_normal = mode == GGML_ROPE_TYPE_NORMAL;1059 const bool is_neox = mode == GGML_ROPE_TYPE_NEOX;1060 const bool is_imrope = mode == GGML_ROPE_TYPE_IMROPE;1061 const bool zero_view_offset = op->src[0]->view_src == nullptr || op->src[0]->view_offs == 0;1062 const bool has_sections = ggml_get_op_params_i32(op, 11) > 0 || ggml_get_op_params_i32(op, 12) > 0 ||1063 ggml_get_op_params_i32(op, 13) > 0;1064 1065 supported =1066 zero_view_offset && ndims <= 512 &&1067 (is_normal || (is_neox && ndims % 16 == 0) || (is_imrope && ndims % 16 == 0 && has_sections));1068 } else {1069 supported = false;1070 }1071 break;1072 case GGML_OP_RMS_NORM:1073 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1074 op->ne[0] % 16 == 0 && ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0]);1075 break;1076 case GGML_OP_NORM:1077 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1078 op->ne[0] % 16 == 0 && ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0]);1079 break;1080 case GGML_OP_L2_NORM:1081 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1082 op->ne[0] % 16 == 0 && ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0]);1083 break;1084 case GGML_OP_GROUP_NORM:1085 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1086 ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0]) &&1087 ggml_get_op_params_i32(op, 0) > 0;1088 break;1089 case GGML_OP_IM2COL:1090 supported = op->src[0] && op->src[1] &&1091 ((op->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32) ||1092 (op->type == GGML_TYPE_F16 &&1093 (op->src[1]->type == GGML_TYPE_F16 || op->src[1]->type == GGML_TYPE_F32))) &&1094 ggml_is_contiguous(op) && ggml_is_contiguous(op->src[1]) &&1095 op->nb[0] == ggml_type_size(op->type) && op->src[1]->nb[0] == ggml_type_size(op->src[1]->type);1096 break;1097 case GGML_OP_CONV_2D:1098 {1099 // First-cut conv_2d_f32_me kernel constraints. Anything outside1100 // this falls back to CPU (it's a strict subset on purpose).1101 if (!op->src[0] || !op->src[1]) {1102 supported = false;1103 break;1104 }1105 if (op->type != GGML_TYPE_F32 || op->src[0]->type != GGML_TYPE_F32 ||1106 op->src[1]->type != GGML_TYPE_F32) {1107 supported = false;1108 break;1109 }1110 if (!ggml_is_contiguous(op) || !ggml_is_contiguous(op->src[0]) || !ggml_is_contiguous(op->src[1])) {1111 supported = false;1112 break;1113 }1114 1115 const ggml_tensor * flt = op->src[0]; // [Kw, Kh, Cin, Cout]1116 const ggml_tensor * in = op->src[1]; // [W, H, Cin, N]1117 const int32_t s0 = ggml_get_op_params_i32(op, 0);1118 const int32_t s1 = ggml_get_op_params_i32(op, 1);1119 const int32_t p0 = ggml_get_op_params_i32(op, 2);1120 const int32_t p1 = ggml_get_op_params_i32(op, 3);1121 const int32_t d0 = ggml_get_op_params_i32(op, 4);1122 const int32_t d1 = ggml_get_op_params_i32(op, 5);1123 1124 const int64_t Kw = flt->ne[0];1125 const int64_t Kh = flt->ne[1];1126 const int64_t Cin = flt->ne[2];1127 const int64_t Cout = flt->ne[3];1128 const int64_t H = in->ne[1];1129 (void) in->ne[0];1130 1131 if (s0 < 1 || s1 < 1 || !(d0 == 1 && d1 == 1) || Cin % 16 != 0 || Cout % 16 != 0 || in->ne[3] != 1) {1132 supported = false;1133 break;1134 }1135 const int64_t OW = op->ne[0];1136 const int64_t OH = op->ne[1];1137 if (OW <= 0 || OH <= 0) {1138 supported = false;1139 break;1140 }1141 (void) p0;1142 (void) p1;1143 1144 // Mirror the kernel's sizing:1145 // if K_TILES * per_KT_bytes <= budget: 1 buffer, n_chunks=11146 // else: 2 buffers (double-buffer), shrink chunk_KT until1147 // 2*chunk_KT*per_KT_bytes <= budget.1148 const int64_t Hp = H + 2 * p1;1149 const int64_t OW_pad = (OW + 15) & ~15;1150 const int64_t Wp_a = OW_pad;1151 const bool need_stage = (OW % 16 != 0);1152 const int64_t stage_bytes = need_stage ? (Cout * OH * OW_pad * 4) : 0;1153 const int64_t L2SCP_BUDGET = 1500 * 1024;1154 // Per-hart partial-TenC scratch (mirrors kernel MAX_TILES_PER_HART=2):1155 // 32 minions x 2 tiles x 1024 bytes = 64 KB per shire.1156 const int64_t scratch_bytes = 32 * 2 * 16 * 16 * 4;1157 const int64_t budget = L2SCP_BUDGET - stage_bytes - scratch_bytes;1158 const int64_t per_KT_bytes = Kh * Kw * Cout * 16 * 4 + Kw * 16 * Hp * Wp_a * 4;1159 const int64_t K_TILES = Cin / 16;1160 1161 int64_t chunk_KT_calc;1162 int64_t n_chunks_calc;1163 if (K_TILES * per_KT_bytes <= budget) {1164 chunk_KT_calc = K_TILES;1165 n_chunks_calc = 1;1166 } else {1167 chunk_KT_calc = K_TILES;1168 while (chunk_KT_calc > 1 && 2 * chunk_KT_calc * per_KT_bytes > budget) {1169 chunk_KT_calc--;1170 }1171 while (chunk_KT_calc > 1 && K_TILES % chunk_KT_calc != 0) {1172 chunk_KT_calc--;1173 }1174 if (chunk_KT_calc < 1) {1175 supported = false;1176 break;1177 }1178 n_chunks_calc = K_TILES / chunk_KT_calc;1179 }1180 1181 if (n_chunks_calc > 1) {1182 const int64_t M_TILES = Cout / 16;1183 const int64_t w_tiles = (OW + 15) / 16;1184 const int64_t total_tiles = OH * w_tiles * M_TILES;1185 // MAX_TILES_PER_HART = 2 (mirrors kernel constant).1186 const int64_t max_workers = (need_stage ? 32 : 1024) * 2;1187 if (total_tiles > max_workers) {1188 supported = false;1189 break;1190 }1191 }1192 1193 supported = true;1194 break;1195 }1196 case GGML_OP_SCALE:1197 // F32 contiguous, total elements must be cache line aligned (16 floats)1198 supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1199 ggml_is_contiguous(op) && ggml_is_contiguous(op->src[0]) && (ggml_nelements(op) % 16 == 0);1200 break;