Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
ggml-et.cpp1877 linesDownload Raw Back to ggml-et
1#include "ggml-et.h"2 3#include "ggml-backend-impl.h"4#include "ggml-backend.h"5#include "ggml-et-common.h"6#include "ggml-et-kernels.h"7#include "ggml-et-memops.h"8#include "ggml-et-ops.h"9#include "ggml-impl.h"10#include "ggml.h"11 12#include <stdarg.h>13 14#include <cstdarg>15#include <cstdio>16#include <cstdlib>17#include <cstring>18#include <vector>19 20#if __has_include(<filesystem>)21#    include <filesystem>22namespace fs = std::filesystem;23#elif __has_include(<experimental/filesystem>)24#    include <experimental/filesystem>25namespace fs = std::experimental::filesystem;26#else27#    error "cannot include the filesystem library"28#endif29 30/*31 * ggml_et_dump_tensor_metadata32 * @brief prints the metadata of a single tensorf33 */34static void ggml_et_dump_tensor_metadata(const ggml_tensor * ggtensor, size_t indent_level, const char * title) {35    char * spaces = (char *) alloca(indent_level + 1);36    memset(spaces, ' ', indent_level);37    spaces[indent_level] = '\0';38    fprintf(stderr,39            "%s%s: %s\n"40            "%s  type: %s\n"41            "%s  ne: %lld %lld %lld %lld\n"42            "%s  nb: %zu %zu %zu %zu\n"43            "%s  op: %s\n"44            "%s  data: %p\n"45            "%s  src0: %p\n",46            spaces, title, ggtensor->name, spaces, ggml_type_name(ggtensor->type), spaces, (long long) ggtensor->ne[0],47            (long long) ggtensor->ne[1], (long long) ggtensor->ne[2], (long long) ggtensor->ne[3], spaces,48            ggtensor->nb[0], ggtensor->nb[1], ggtensor->nb[2], ggtensor->nb[3], spaces, ggml_op_name(ggtensor->op),49            spaces, ggtensor->data, spaces, (void *) ggtensor->src[0]);50}51 52/*53 * ggml_et_dump_operator_metadata54 * @brief prints the metadata of a single tensor (or operator) including it's input and views55 */56static void ggml_et_dump_operator_metadata(const ggml_tensor * ggtensor) {57    GGML_ASSERT(ggtensor != NULL);58    ggml_et_dump_tensor_metadata(ggtensor, 0, "GGML tensor");59    for (int i = 0; i < GGML_MAX_SRC && ggtensor->src[i]; i++) {60        char arr[16];61        int  n = snprintf(arr, sizeof(arr), "src[%i]->name", i);62        GGML_ASSERT((unsigned) n < sizeof(arr) && "printed too much data to stack buffer");63        ggml_et_dump_tensor_metadata(ggtensor->src[i], 2, arr);64    }65    if (ggtensor->view_src) {66        ggml_et_dump_tensor_metadata(ggtensor, 2, "view_src");67    }68}69 70static struct ggml_et_driver {71    std::shared_ptr<dev::IDeviceLayer>                device_layer;72    std::shared_ptr<rt::IRuntime>                     runtime;73    std::unique_ptr<std::ofstream>                    profile_stream;74    std::unique_ptr<std::ofstream>                    kernel_id_stream;75    std::vector<std::pair<std::string, rt::KernelId>> kernel_map;76    bool                                              profiling_enabled = false;77} _drv;78 79// Check at runtime environment variables for paths likely holding ET toolchain with sysemu elf files80static std::string ggml_et_get_default_et_path() {81    // List of environment variables to check in order of preference82    const char * const env_vars[] = { "ET_TOOLCHAIN", "TOOLCHAIN_ROOT" };83 84    for (const char * var : env_vars) {85        if (const char * et_path = std::getenv(var)) {86            if (et_path && *et_path != '\0') {87                return fs::path(et_path).string();88            }89        }90    }91 92    // Otherwise assume default93    return fs::path("/opt/et").string();94}95 96// config when using sysemu instead of PCIe hardware device97// adapted from `ainekko/et-platform/esperanto-tools-libs/tools/src/bench.cpp`98static inline auto ggml_et_get_default_sysemu_options() {99    constexpr uint64_t kSysEmuMaxCycles        = std::numeric_limits<uint64_t>::max();100    constexpr uint64_t kSysEmuMinionShiresMask = 0x1FFFFFFFFu;101    const std::string  et_path                 = ggml_et_get_default_et_path() + "/";102 103    emu::SysEmuOptions sysEmuOptions;104 105    // Construct all paths106    sysEmuOptions.bootromTrampolineToBL2ElfPath =107        et_path + "lib/esperanto-fw/BootromTrampolineToBL2/BootromTrampolineToBL2.elf";108    sysEmuOptions.spBL2ElfPath =109        et_path + "lib/esperanto-fw/ServiceProcessorBL2/fast-boot/ServiceProcessorBL2_fast-boot.elf";110    sysEmuOptions.machineMinionElfPath = et_path + "lib/esperanto-fw/MachineMinion/MachineMinion.elf";111    sysEmuOptions.masterMinionElfPath  = et_path + "lib/esperanto-fw/MasterMinion/MasterMinion.elf";112    sysEmuOptions.workerMinionElfPath  = et_path + "lib/esperanto-fw/WorkerMinion/WorkerMinion.elf";113    sysEmuOptions.executablePath       = et_path + "bin/sys_emu";114 115    // Check that each path has a valid existing non-zero file otherwise emulator just silently hangs116    const std::vector<std::string> required_files = {117        sysEmuOptions.bootromTrampolineToBL2ElfPath, sysEmuOptions.spBL2ElfPath,118        sysEmuOptions.machineMinionElfPath,          sysEmuOptions.masterMinionElfPath,119        sysEmuOptions.workerMinionElfPath,           sysEmuOptions.executablePath,120    };121 122    for (const auto & file : required_files) {123        if (!fs::exists(file) || fs::file_size(file) == 0) {124            // Check that each path has a valid existing non-zero file otherwise emulator just silently hangs125            GGML_LOG_ERROR("ET: Unable to find required sysemu file: %s", file.c_str());126            GGML_LOG_ERROR("ET: Confirm et-platform is correctly installed at configured path.");127            abort();128        }129    }130 131    sysEmuOptions.runDir           = (fs::current_path().string() + "/");132    sysEmuOptions.maxCycles        = kSysEmuMaxCycles;133    sysEmuOptions.minionShiresMask = kSysEmuMinionShiresMask;134    sysEmuOptions.puUart0Path      = sysEmuOptions.runDir + "pu_uart0_tx.log";135    sysEmuOptions.puUart1Path      = sysEmuOptions.runDir + "pu_uart1_tx.log";136    sysEmuOptions.spUart0Path      = sysEmuOptions.runDir + "spio_uart0_tx.log";137    sysEmuOptions.spUart1Path      = sysEmuOptions.runDir + "spio_uart1_tx.log";138    sysEmuOptions.startGdb         = false;139    sysEmuOptions.memcheck         = false;140 141    return sysEmuOptions;142}143 144// Forward declaration145static void ggml_et_driver_cleanup();146 147static bool ggml_et_driver_init() {148    if (_drv.runtime != nullptr) {149        assert(_drv.device_layer != nullptr);150    } else {151        try {152#if defined GGML_ET_SYSEMU && GGML_ET_SYSEMU153            // For emulator device using sysEmuOptions provided by function above enabled compiling with `-DGGML_ET_SYSEMU=ON`154            _drv.device_layer = dev::IDeviceLayer::createSysEmuDeviceLayer(ggml_et_get_default_sysemu_options());155#else156            // For physical PCIe device157            _drv.device_layer = dev::IDeviceLayer::createPcieDeviceLayer();158#endif  // GGML_ET_SYSEMU159 160            _drv.runtime = rt::IRuntime::create(_drv.device_layer);161 162            // Initialize profiler if requested via environment variable163            const char * profile_path = getenv("GGML_ET_PROFILE");164            if (profile_path) {165                std::string output_path    = std::string(profile_path) + "/et_runtime_trace.json";166                std::string kernel_id_path = std::string(profile_path) + "/kernel_id.json";167 168                _drv.profile_stream   = std::make_unique<std::ofstream>(output_path);169                _drv.kernel_id_stream = std::make_unique<std::ofstream>(kernel_id_path);170                if (!_drv.profile_stream->is_open()) {171                    GGML_LOG_ERROR("ET: Failed to open profiling output file: %s", output_path.c_str());172                    abort();173                }174                if (!_drv.kernel_id_stream->is_open()) {175                    GGML_LOG_ERROR("ET: Failed to open profiling kernel map: %s", kernel_id_path.c_str());176                    abort();177                }178 179                auto * profiler = _drv.runtime->getProfiler();180                profiler->start(*_drv.profile_stream, rt::IProfiler::OutputType::Json);181                _drv.profiling_enabled = true;182                GGML_LOG_INFO("ET: Runtime profiler started (JSON format)");183 184                // Register cleanup at program exit185                std::atexit(ggml_et_driver_cleanup);186            }187        } catch (const std::exception & e) {188            GGML_LOG_ERROR("ggml_et: %s", e.what());189            if (_drv.device_layer != nullptr) {190                _drv.device_layer.reset();191            }192            if (_drv.runtime != nullptr) {193                _drv.runtime.reset();194            }195            return false;196        }197    }198    return true;199}200 201static std::shared_ptr<dev::IDeviceLayer> ggml_et_devicelayer() {202    return _drv.device_layer;203}204 205std::shared_ptr<rt::IRuntime> ggml_et_runtime() {206    return _drv.runtime;207}208 209static void ggml_et_driver_cleanup() {210    if (_drv.profiling_enabled && _drv.runtime) {211        GGML_LOG_INFO("ET: Stopping runtime profiler");212        auto * profiler = _drv.runtime->getProfiler();213        profiler->stop();214        _drv.profiling_enabled = false;215 216        if (_drv.profile_stream) {217            _drv.profile_stream->close();218            _drv.profile_stream.reset();219        }220 221        // Save kernel map222        if (_drv.kernel_id_stream && !_drv.kernel_map.empty()) {223            auto & os = *_drv.kernel_id_stream;224            // XXX: Manual JSON construction. Not pretty but removes dependency225            os << "{\n";226            for (size_t i = 0; i < _drv.kernel_map.size(); i++) {227                os << "  \"" << _drv.kernel_map[i].first << "\": " << (int) _drv.kernel_map[i].second;228                if (i + 1 < _drv.kernel_map.size()) {229                    os << ",";230                }231                os << "\n";232            }233            os << "}\n";234            _drv.kernel_id_stream->close();235            _drv.kernel_id_stream.reset();236        }237    }238}239 240static ggml_backend_dev_t ggml_backend_et_reg_get_device(ggml_backend_reg_t reg, size_t devidx);241 242static void ggml_backend_et_buffer_free_buffer(ggml_backend_buffer_t buffer) {243    ggml_backend_et_buffer_context * ctx = (ggml_backend_et_buffer_context *) buffer->context;244    if (ctx->data != nullptr) {245        std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();246        if (runtime) {247            runtime->freeDevice(ctx->rtid, static_cast<std::byte *>(ctx->data));248        }249    }250    delete ctx;251}252 253static void * ggml_backend_et_buffer_get_base(ggml_backend_buffer_t buffer) {254    ggml_backend_et_buffer_context * ctx = (ggml_backend_et_buffer_context *) buffer->context;255    return ctx->data;256}257 258static ggml_status ggml_backend_et_buffer_init_tensor(ggml_backend_buffer_t buffer, ggml_tensor * tensor) {259    // View tensors share buffer with their view_src, no additional initialization needed260    if (tensor->view_src != NULL) {261        return GGML_STATUS_SUCCESS;262    }263 264    const size_t original_size = ggml_nbytes(tensor);265    const size_t padded_size   = ggml_backend_buft_get_alloc_size(buffer->buft, tensor);266 267    // Clear padding bytes to avoid NaN values268    // XXX: Martin - do we need this?269    if (padded_size > original_size) {270        const size_t padding_size = padded_size - original_size;271 272        // Get device context to access memops kernel273        ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buffer->buft->device->context;274        if (!dev_ctx) {275            GGML_LOG_ERROR("ET: Failed to get device context for padding clear");276            return GGML_STATUS_FAILED;277        }278 279        // Use device-side memset kernel for efficient padding clear280        std::byte * padding_ptr = static_cast<std::byte *>(tensor->data) + original_size;281        if (!ggml_et_memset(dev_ctx, padding_ptr, 0, padding_size)) {282            GGML_LOG_ERROR("ET: Failed to clear padding using memset kernel for tensor %s", tensor->name);283            return GGML_STATUS_FAILED;284        }285    }286 287    return GGML_STATUS_SUCCESS;288}289 290static void ggml_backend_et_buffer_set_tensor(ggml_backend_buffer_t buffer,291                                              ggml_tensor *         tensor,292                                              const void *          data,293                                              size_t                offset,294                                              size_t                size) {295    std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();296    if (!runtime) {297        return;298    }299 300    // Create short-lived stream for this transfer301    ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buffer->buft->device->context;302    rt::StreamId                     stream  = dev_ctx->default_stream;303 304    std::byte *       dst_ptr = static_cast<std::byte *>(tensor->data) + offset;305    const std::byte * src_ptr = static_cast<const std::byte *>(data);306 307    rt::EventId event = runtime->memcpyHostToDevice(stream, src_ptr, dst_ptr, size, true /*barrier*/);308 309    runtime->waitForEvent(event);310}311 312static void ggml_backend_et_buffer_get_tensor(ggml_backend_buffer_t buffer,313                                              const ggml_tensor *   tensor,314                                              void *                data,315                                              size_t                offset,316                                              size_t                size) {317    std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();318    if (!runtime) {319        return;320    }321 322    ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buffer->buft->device->context;323    rt::StreamId                     stream  = dev_ctx->default_stream;324 325    const std::byte * src_ptr = static_cast<const std::byte *>(tensor->data) + offset;326    std::byte *       dst_ptr = static_cast<std::byte *>(data);327 328    rt::EventId event = runtime->memcpyDeviceToHost(stream, src_ptr, dst_ptr, size, true /*barrier*/);329 330    runtime->waitForEvent(event);331}332 333static bool ggml_backend_et_buffer_cpy_tensor(ggml_backend_buffer_t buffer,334                                              const ggml_tensor *   src,335                                              ggml_tensor *         dst) {336    GGML_UNUSED(buffer);337    GGML_UNUSED(src);338    GGML_UNUSED(dst);339    return false;340}341 342static void ggml_backend_et_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) {343    ggml_backend_et_buffer_context * ctx = (ggml_backend_et_buffer_context *) buffer->context;344 345    if (ctx->size == 0 || ctx->data == nullptr) {346        return;347    }348 349    // Get device context to access memops kernel350    ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buffer->buft->device->context;351    if (!dev_ctx) {352        GGML_LOG_ERROR("ET: Failed to get device context for buffer clear");353        return;354    }355 356    // Use device-side memset kernel for efficient clearing357    if (!ggml_et_memset(dev_ctx, ctx->data, value, ctx->size)) {358        GGML_LOG_ERROR("ET: buffer_clear failed using memset kernel");359        return;360    }361 362    GGML_LOG_DEBUG("ET: Buffer cleared successfully using memops kernel");363}364 365static const struct ggml_backend_buffer_i ggml_backend_et_buffer_i = {366    /* .free_buffer     = */ ggml_backend_et_buffer_free_buffer,367    /* .get_base        = */ ggml_backend_et_buffer_get_base,368    /* .init_tensor     = */ ggml_backend_et_buffer_init_tensor,369    /* .memset_tensor   = */ NULL,370    /* .set_tensor      = */ ggml_backend_et_buffer_set_tensor,371    /* .get_tensor      = */ ggml_backend_et_buffer_get_tensor,372    /* .set_tensor_2d   = */ NULL,373    /* .get_tensor_2d   = */ NULL,374    /* .cpy_tensor      = */ ggml_backend_et_buffer_cpy_tensor,375    /* .clear           = */ ggml_backend_et_buffer_clear,376    /* .reset           = */ NULL,377};378 379static const char * ggml_backend_et_buffer_type_get_name(ggml_backend_buffer_type_t buft) {380    GGML_UNUSED(buft);381    return GGML_ET_NAME;382}383 384static ggml_backend_buffer_t ggml_backend_et_buffer_type_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) {385    ggml_backend_et_buffer_type_context * btctx = (ggml_backend_et_buffer_type_context *) buft->context;386 387    ggml_backend_et_buffer_context * ctx = new ggml_backend_et_buffer_context;388    ctx->devidx                          = btctx->devidx;389    ctx->size                            = size;390 391    std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();392    if (!runtime) {393        delete ctx;394        return nullptr;395    }396 397    std::vector<rt::DeviceId> rtids = runtime->getDevices();398    if (static_cast<size_t>(btctx->devidx) >= rtids.size()) {399        delete ctx;400        return nullptr;401    }402    ctx->rtid = rtids[btctx->devidx];403 404    ctx->data = runtime->mallocDevice(ctx->rtid, size);405    if (ctx->data == nullptr) {406        delete ctx;407        return nullptr;408    }409 410    return ggml_backend_buffer_init(buft, ggml_backend_et_buffer_i, ctx, size);411}412 413static size_t ggml_backend_et_buffer_type_get_alignment(ggml_backend_buffer_type_t buft) {414    std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();415    if (!runtime || !buft->device) {416        return GGML_MEM_ALIGN;417    }418 419    ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buft->device->context;420    rt::DeviceProperties             prop    = runtime->getDeviceProperties(dev_ctx->rtid);421    return prop.cacheLineSize_;422}423 424static size_t ggml_backend_et_buffer_type_get_max_size(ggml_backend_buffer_type_t buft) {425    if (buft->device) {426        ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) buft->device->context;427        return dev_ctx->total_mem;428    }429    return SIZE_MAX;430}431 432static size_t ggml_backend_et_buffer_type_get_alloc_size(ggml_backend_buffer_type_t buft, const ggml_tensor * tensor) {433    GGML_UNUSED(buft);434    return ggml_nbytes_pad(tensor);435}436 437static bool ggml_backend_et_buffer_type_is_host(ggml_backend_buffer_type_t buft) {438    GGML_UNUSED(buft);439    return false;440}441 442static const struct ggml_backend_buffer_type_i ggml_backend_et_buffer_type_i = {443    /* .get_name         = */ ggml_backend_et_buffer_type_get_name,444    /* .alloc_buffer     = */ ggml_backend_et_buffer_type_alloc_buffer,445    /* .get_alignment    = */ ggml_backend_et_buffer_type_get_alignment,446    /* .get_max_size     = */ ggml_backend_et_buffer_type_get_max_size,447    /* .get_alloc_size   = */ ggml_backend_et_buffer_type_get_alloc_size,448    /* .is_host          = */ ggml_backend_et_buffer_type_is_host,449};450 451static const char * ggml_backend_et_get_name(ggml_backend_t backend) {452    GGML_UNUSED(backend);453    return GGML_ET_NAME;454}455 456static void ggml_backend_et_free(ggml_backend_t backend) {457    ggml_backend_et_context *     et_ctx  = (ggml_backend_et_context *) backend->context;458    std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();459 460    // Clean up kernels on this device before freeing backend461    ggml_backend_dev_t dev = ggml_backend_et_reg_get_device(ggml_backend_et_reg(), et_ctx->devidx);462    if (dev && dev->context) {463        ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) dev->context;464 465        if (_drv.profiling_enabled) {466            auto kernels = ggml_et_get_loaded_kernels(dev_ctx);467            _drv.kernel_map.insert(_drv.kernel_map.end(), kernels.begin(), kernels.end());468        }469 470        ggml_et_unload_all_kernels(dev_ctx);471 472        if (runtime) {473            if (dev_ctx->trace_buffer) {474                runtime->freeDevice(dev_ctx->rtid, dev_ctx->trace_buffer);475                dev_ctx->trace_buffer = nullptr;476            }477            // Drain any in-flight uberkernel launches before freeing the478            // device buffers they read from.479            runtime->waitForStream(dev_ctx->default_stream);480            for (auto & slot : dev_ctx->uberkernel.slots) {481                if (slot.device_insts) {482                    runtime->freeDevice(dev_ctx->rtid, slot.device_insts);483                    slot.device_insts = nullptr;484                }485                if (slot.device_params) {486                    runtime->freeDevice(dev_ctx->rtid, slot.device_params);487                    slot.device_params = nullptr;488                }489                slot.has_pending = false;490            }491        }492    }493 494    delete et_ctx;495    delete backend;496}497 498static ggml_backend_buffer_type_t ggml_backend_et_get_default_buffer_type(ggml_backend_t backend) {499    ggml_backend_et_context * et_ctx = (ggml_backend_et_context *) backend->context;500 501    return ggml_backend_et_buffer_type(et_ctx->devidx);502}503 504static void ggml_backend_et_set_tensor_async(ggml_backend_t backend,505                                             ggml_tensor *  tensor,506                                             const void *   data,507                                             size_t         offset,508                                             size_t         size) {509    std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();510    if (!runtime) {511        return;512    }513 514    ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) backend->device->context;515    rt::StreamId                     stream  = dev_ctx->default_stream;516 517    std::byte *       dst_ptr = static_cast<std::byte *>(tensor->data) + offset;518    const std::byte * src_ptr = static_cast<const std::byte *>(data);519 520    runtime->memcpyHostToDevice(stream, src_ptr, dst_ptr, size, true /*barrier*/);521}522 523static void ggml_backend_et_get_tensor_async(ggml_backend_t      backend,524                                             const ggml_tensor * tensor,525                                             void *              data,526                                             size_t              offset,527                                             size_t              size) {528    std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();529    if (!runtime) {530        return;531    }532 533    ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) backend->device->context;534    rt::StreamId                     stream  = dev_ctx->default_stream;535 536    const std::byte * src_ptr = static_cast<const std::byte *>(tensor->data) + offset;537    std::byte *       dst_ptr = static_cast<std::byte *>(data);538 539    runtime->memcpyDeviceToHost(stream, src_ptr, dst_ptr, size, true /*barrier*/);540}541 542static bool ggml_backend_et_cpy_tensor_async(ggml_backend_t      backend_src,543                                             ggml_backend_t      backend_dst,544                                             const ggml_tensor * src,545                                             ggml_tensor *       dst) {546    GGML_UNUSED(backend_src);547    GGML_UNUSED(backend_dst);548    GGML_UNUSED(src);549    GGML_UNUSED(dst);550    return false;551}552 553static void ggml_backend_et_synchronize(ggml_backend_t backend) {554    std::shared_ptr<rt::IRuntime> runtime = ggml_et_runtime();555    if (!runtime) {556        return;557    }558 559    ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) backend->device->context;560    runtime->waitForStream(dev_ctx->default_stream);561 562    auto errors = runtime->retrieveStreamErrors(dev_ctx->default_stream);563    if (errors.empty()) {564        return;565    }566    for (const auto & err : errors) {567        GGML_LOG_ERROR("ET: stream error detected at synchronization point. Code: %d,Type: %d\n", (int) err.errorCode_,568                       (int) err.errorContext_.value()[0].type_);569    }570    abort();571}572 573static bool ggml_et_can_fuse(const ggml_cgraph * cgraph, int node_idx, std::initializer_list<ggml_op> ops) {574    if (!ggml_can_fuse(cgraph, node_idx, ops)) {575        return false;576    }577 578    if (ops.size() == 2 && ops.begin()[0] == GGML_OP_MUL_MAT && ops.begin()[1] == GGML_OP_ADD) {579        const ggml_tensor * mm  = cgraph->nodes[node_idx];580        const ggml_tensor * add = cgraph->nodes[node_idx + 1];581 582        // Only Q8_0 weights x F32 activations -> F32 (the kernel that has583        // the bias path).  Other MM variants must wait for their own kernel584        // bias support.585        if (mm->type != GGML_TYPE_F32 || mm->src[0]->type != GGML_TYPE_Q8_0 || mm->src[1]->type != GGML_TYPE_F32) {586            return false;587        }588 589        // ADD must be F32 and one of its operands must be the MM output.590        if (add->type != GGML_TYPE_F32) {591            return false;592        }593        if (add->src[0] != mm && add->src[1] != mm) {594            return false;595        }596 597        const ggml_tensor * bias = (add->src[0] == mm) ? add->src[1] : add->src[0];598 599        if (bias->type != GGML_TYPE_F32) {600            return false;601        }602 603        // No broadcasting: bias shape must equal MM output shape.604        for (int i = 0; i < GGML_MAX_DIMS; ++i) {605            if (bias->ne[i] != mm->ne[i]) {606                return false;607            }608        }609 610        // Bias and dst must be contiguous and have identical strides - the611        // kernel uses dst-style offset arithmetic against bias's nb[].612        if (!ggml_is_contiguous(bias) || !ggml_is_contiguous(mm)) {613            return false;614        }615        for (int i = 0; i < GGML_MAX_DIMS; ++i) {616            if ((int64_t) bias->nb[i] != (int64_t) add->nb[i]) {617                return false;618            }619        }620    }621 622    if (ops.size() == 2 && ops.begin()[0] == GGML_OP_RMS_NORM && ops.begin()[1] == GGML_OP_MUL) {623        const ggml_tensor * rms_norm = cgraph->nodes[node_idx];624        const ggml_tensor * mul      = cgraph->nodes[node_idx + 1];625 626        // ET only supports F32627        if (rms_norm->src[0]->type != GGML_TYPE_F32 || mul->type != GGML_TYPE_F32) {628            return false;629        }630 631        // Identify the weights tensor (the MUL operand that isn't rms_norm output)632        const ggml_tensor * weights = (mul->src[0] == rms_norm) ? mul->src[1] : mul->src[0];633 634        if (weights->type != GGML_TYPE_F32) {635            return false;636        }637 638        // Both inputs must be contiguous (ET hardware requirement)639        if (!ggml_is_contiguous(rms_norm->src[0]) || !ggml_is_contiguous_rows(weights)) {640            return false;641        }642 643        // ET requires cache-aligned rows (ne[0] % 16 == 0)644        if (rms_norm->src[0]->ne[0] % 16 != 0 || weights->ne[0] % 16 != 0) {645            return false;646        }647 648        // Fused kernel doesn't handle dim-0 broadcasting649        if (weights->ne[0] != rms_norm->src[0]->ne[0]) {650            return false;651        }652    }653 654    return true;655}656 657static ggml_status ggml_backend_et_graph_compute(ggml_backend_t backend, ggml_cgraph * cgraph) {658    ggml_backend_et_device_context * dev_ctx = (ggml_backend_et_device_context *) backend->device->context;659    ggml_et_uberkernel_begin_graph(&dev_ctx->uberkernel);660 661    for (int i = 0; i < cgraph->n_nodes; i++) {662        ggml_tensor * node = cgraph->nodes[i];663 664        if (node->op == GGML_OP_NONE || node->op == GGML_OP_VIEW || node->op == GGML_OP_RESHAPE ||665            node->op == GGML_OP_PERMUTE || node->op == GGML_OP_TRANSPOSE) {666            continue;667        }668 669        // --- Fusion checks (before regular dispatch) ---670        if (ggml_et_can_fuse(cgraph, i, { GGML_OP_RMS_NORM, GGML_OP_MUL })) {671            ggml_et_op_rms_norm_mul(dev_ctx, node, cgraph->nodes[i + 1]);672            i++;  // skip the MUL node673            continue;674        }675        if (ggml_et_can_fuse(cgraph, i, { GGML_OP_MUL_MAT, GGML_OP_ADD })) {676            ggml_et_op_mul_mat(dev_ctx, node, cgraph->nodes[i + 1]);677            i++;  // skip the ADD node678            continue;679        }680 681        switch (node->op) {682            case GGML_OP_SQR:683                ggml_et_op_sqr(dev_ctx, node);684                break;685 686            case GGML_OP_UNARY:687                ggml_et_op_unary(dev_ctx, node);688                break;689 690            case GGML_OP_SUM_ROWS:691                ggml_et_op_sum_rows(dev_ctx, node);692                break;693 694            case GGML_OP_MEAN:695                ggml_et_op_mean(dev_ctx, node);696                break;697 698            case GGML_OP_CLAMP:699                ggml_et_op_clamp(dev_ctx, node);700                break;701 702            case GGML_OP_MUL:703                ggml_et_op_mul(dev_ctx, node);704                break;705 706            case GGML_OP_ADD:707                ggml_et_op_add(dev_ctx, node);708                break;709 710            case GGML_OP_SUB:711                ggml_et_op_sub(dev_ctx, node);712                break;713 714            case GGML_OP_CUMSUM:715                ggml_et_op_cumsum(dev_ctx, node);716                break;717 718            case GGML_OP_MUL_MAT:719                ggml_et_op_mul_mat(dev_ctx, node);720                break;721 722            case GGML_OP_MUL_MAT_ID:723                ggml_et_op_mul_mat_id(dev_ctx, node);724                break;725 726            case GGML_OP_ROPE:727                ggml_et_op_rope(dev_ctx, node);728                break;729 730            case GGML_OP_RMS_NORM:731                ggml_et_op_rms_norm(dev_ctx, node);732                break;733 734            case GGML_OP_NORM:735                ggml_et_op_norm(dev_ctx, node);736                break;737 738            case GGML_OP_L2_NORM:739                ggml_et_op_l2_norm(dev_ctx, node);740                break;741 742            case GGML_OP_GROUP_NORM:743                ggml_et_op_group_norm(dev_ctx, node);744                break;745 746            case GGML_OP_SCALE:747                ggml_et_op_scale(dev_ctx, node);748                break;749 750            case GGML_OP_GLU:751                ggml_et_op_glu(dev_ctx, node);752                break;753 754            case GGML_OP_SOFT_MAX:755                ggml_et_op_softmax(dev_ctx, node);756                break;757 758            case GGML_OP_IM2COL:759                ggml_et_op_im2col(dev_ctx, node);760                break;761 762            case GGML_OP_CONV_2D:763                ggml_et_op_conv_2d(dev_ctx, node);764                break;765 766            case GGML_OP_FLASH_ATTN_EXT:767                ggml_et_op_flash_attn_ext(dev_ctx, node);768                break;769 770            case GGML_OP_GET_ROWS:771                ggml_et_op_get_rows(dev_ctx, node);772                break;773 774            case GGML_OP_CONT:775                ggml_et_op_cont(dev_ctx, node);776                break;777 778            case GGML_OP_CPY:779                ggml_et_op_cpy(dev_ctx, node);780                break;781 782            case GGML_OP_CONCAT:783                ggml_et_op_concat(dev_ctx, node);784                break;785 786            case GGML_OP_REPEAT:787                ggml_et_op_repeat(dev_ctx, node);788                break;789 790            case GGML_OP_SSM_CONV:791                ggml_et_op_ssm_conv(dev_ctx, node);792                break;793 794            case GGML_OP_SSM_SCAN:795                ggml_et_op_ssm_scan(dev_ctx, node);796                break;797 798            case GGML_OP_PAD:799                ggml_et_op_pad(dev_ctx, node);800                break;801 802            case GGML_OP_SET_ROWS:803                ggml_et_op_set_rows(dev_ctx, node);804                break;805 806            case GGML_OP_FILL:807                ggml_et_op_fill(dev_ctx, node);808                break;809 810            case GGML_OP_DIAG:811                ggml_et_op_diag(dev_ctx, node);812                break;813 814            case GGML_OP_TRI:815                ggml_et_op_tri(dev_ctx, node);816                break;817 818            case GGML_OP_SOLVE_TRI:819                ggml_et_op_solve_tri(dev_ctx, node);820                break;821 822            case GGML_OP_SET:823                ggml_et_op_set(dev_ctx, node);824                break;825 826            case GGML_OP_RWKV_WKV6:827                ggml_et_op_rwkv_wkv6(dev_ctx, node);828                break;829 830            case GGML_OP_RWKV_WKV7:831                ggml_et_op_rwkv_wkv7(dev_ctx, node);832                break;833 834            case GGML_OP_GATED_DELTA_NET:835                ggml_et_op_gated_delta_net(dev_ctx, node);836                break;837 838            default:839                ggml_et_uberkernel_abort_graph(&dev_ctx->uberkernel);840                GGML_LOG_ERROR("ET: Unsupported operation in graph: %s", ggml_op_name(node->op));841                return GGML_STATUS_FAILED;842        }843 844        if (ggml_et_uberkernel_failed(&dev_ctx->uberkernel)) {845            ggml_et_uberkernel_abort_graph(&dev_ctx->uberkernel);846            return GGML_STATUS_FAILED;847        }848    }849 850    if (!ggml_et_uberkernel_end_graph(dev_ctx)) {851        ggml_et_uberkernel_abort_graph(&dev_ctx->uberkernel);852        return GGML_STATUS_FAILED;853    }854 855    return GGML_STATUS_SUCCESS;856}857 858// Check that elements within each row are contiguous (nb[0] == type_size).859// Higher-dim strides can be arbitrary - kernels navigate them via byte offsets.860static bool et_ggml_is_row_contiguous(const ggml_tensor * t) {861    return t->nb[0] == ggml_type_size(t->type);862}863 864static bool ggml_backend_et_device_supports_op(ggml_backend_dev_t dev, const ggml_tensor * op) {865    GGML_UNUSED(dev);866 867    bool supported = false;868    switch (op->op) {869        case GGML_OP_CUMSUM:870            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&871                        op->src[0]->nb[0] == sizeof(float) && ggml_is_contiguous(op);872            break;873        case GGML_OP_SQR:874            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&875                        op->ne[0] % 16 == 0 && ggml_is_contiguous(op) && ggml_is_contiguous(op->src[0]);876            break;877        case GGML_OP_SUM_ROWS:878            // dst has ne[0]=1, src0 row length must be cache-aligned879            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&880                        op->src[0]->ne[0] % 16 == 0 && ggml_is_contiguous(op->src[0]);881            break;882        case GGML_OP_MEAN:883            // Kernel handles arbitrary ne00 (per-row alignment guard with884            // scalar tail), so no row-length divisibility constraint here.885            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&886                        ggml_is_contiguous(op->src[0]);887            break;888        case GGML_OP_CLAMP:889            // Element-wise; kernel distributes by cache lines and handles a890            // scalar tail, so any contiguous F32 size is fine - including the891            // 1x1x1x1 scalar case.892            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&893                        ggml_is_contiguous(op) && ggml_is_contiguous(op->src[0]);894            break;895        case GGML_OP_UNARY:896            // Only require dim-0 contiguity (nb[0] == sizeof(float)). Higher897            // dims may be arbitrarily strided views; the kernel walks per-row898            // using all four nb[] values. See unary_f32.c entry_point.899            if (op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&900                ggml_nelements(op) % 16 == 0 && op->nb[0] == sizeof(float) && op->src[0]->nb[0] == sizeof(float)) {901                switch (ggml_get_unary_op(op)) {902                    case GGML_UNARY_OP_ABS:903                    case GGML_UNARY_OP_SGN:904                    case GGML_UNARY_OP_NEG:905                    case GGML_UNARY_OP_STEP:906                    case GGML_UNARY_OP_TANH:907                    case GGML_UNARY_OP_ELU:908                    case GGML_UNARY_OP_RELU:909                    case GGML_UNARY_OP_SIGMOID:910                    case GGML_UNARY_OP_GELU:911                    case GGML_UNARY_OP_GELU_QUICK:912                    case GGML_UNARY_OP_SILU:913                    case GGML_UNARY_OP_HARDSWISH:914                    case GGML_UNARY_OP_HARDSIGMOID:915                    case GGML_UNARY_OP_EXP:916                    case GGML_UNARY_OP_EXPM1:917                    case GGML_UNARY_OP_SOFTPLUS:918                    case GGML_UNARY_OP_GELU_ERF:919                    case GGML_UNARY_OP_FLOOR:920                    case GGML_UNARY_OP_CEIL:921                    case GGML_UNARY_OP_ROUND:922                    case GGML_UNARY_OP_TRUNC:923                        supported = true;924                        break;925                    default:926                        break;927                }928            }929            break;930        case GGML_OP_MUL:931        case GGML_OP_ADD:932        case GGML_OP_SUB:933            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 && op->src[1] &&934                        op->src[1]->type == GGML_TYPE_F32 && op->nb[0] == sizeof(float) &&935                        op->src[0]->nb[0] == sizeof(float) &&936                        (op->src[1]->nb[0] == sizeof(float) || op->src[1]->ne[0] == 1) &&937                        op->nb[1] == op->ne[0] * sizeof(float);938            break;939        case GGML_OP_MUL_MAT:940            // Support Q8_0 x F32 -> F32, F16 x F32 -> F32, F16 x F16 -> F32, and F32 x F32 -> F32 matrix multiplication941            // Stride requirements: first dimension must be contiguous for all tensors942            if (op->type == GGML_TYPE_F32 &&943                ((op->src[0]->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32) ||944                 (op->src[0]->type == GGML_TYPE_F16 && op->src[1]->type == GGML_TYPE_F16)) &&945                op->ne[0] % 16 == 0 &&          // dst row length for tensor-store path946                op->src[0]->ne[1] % 16 == 0 &&  // m947                op->src[0]->ne[0] % 16 == 0 &&  // k948                ggml_is_contiguous(op->src[0]) && ggml_is_contiguous(op->src[1])) {949                // Special path for the FP32 TensorFMA kernel950                // Limitation - generic kernels can tolerate non-cache-aligned dst rows951                // because they publish each output element atomically. The matrix952                // engine path still uses tiled tensor stores, so keep dst rows aligned.953                // The m edge is difficult to do because of the 4 conseqtive load hardware limitation954                // And the k edge is impossible because that is encoded as `stride & 0xFFFFFFFFFFC0ULL` which becomes 0 for stride 16 (4x FP32) :(955                // FIXME: Right now this overwrites the mul_mat_f32 kernel - whatever. Fix later. Demo code956                supported = true;957            } else if (op->type == GGML_TYPE_F32 && op->src[0] &&958                       (op->src[0]->type == GGML_TYPE_F16 || op->src[0]->type == GGML_TYPE_F32) && op->src[1] &&959                       (op->src[1]->type == GGML_TYPE_F16 || op->src[1]->type == GGML_TYPE_F32)) {960                // Check first dimension contiguity requirements961                bool src0_first_dim_contiguous = (op->src[0]->nb[0] == ggml_type_size(op->src[0]->type));962                bool src1_first_dim_contiguous = (op->src[1]->nb[0] == ggml_type_size(op->src[1]->type));963                bool dst_first_dim_contiguous  = (op->nb[0] == sizeof(float));964 965                // Check destination stride ordering (only for dimensions with ne > 1)966                bool dst_properly_ordered = true;967                for (int d = 0; d < 3; d++) {968                    if (op->ne[d] > 1 && op->ne[d + 1] > 1 && op->nb[d] > op->nb[d + 1]) {969                        dst_properly_ordered = false;970                    }971                }972 973                supported = src0_first_dim_contiguous && src1_first_dim_contiguous && dst_first_dim_contiguous &&974                            dst_properly_ordered;975            } else if (op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_Q8_0 && op->src[1] &&976                       op->src[1]->type == GGML_TYPE_F32) {977                // Keep the existing quantized path constraints separate from the978                // relaxed non-quant generic fallback.979                bool src0_first_dim_contiguous = (op->src[0]->nb[0] == ggml_type_size(op->src[0]->type));980                bool src1_first_dim_contiguous = (op->src[1]->nb[0] == ggml_type_size(op->src[1]->type));981                bool dst_first_dim_contiguous  = (op->nb[0] == sizeof(float));982 983                bool dst_properly_ordered = true;984                for (int d = 0; d < 3; d++) {985                    if (op->ne[d] > 1 && op->ne[d + 1] > 1 && op->nb[d] > op->nb[d + 1]) {986                        dst_properly_ordered = false;987                    }988                }989 990                supported = src0_first_dim_contiguous && src1_first_dim_contiguous && dst_first_dim_contiguous &&991                            dst_properly_ordered;992 993            } else if (op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_Q4_0 && op->src[1] &&994                       op->src[1]->type == GGML_TYPE_F32) {995                // Keep the existing quantized path constraints separate from the996                // relaxed non-quant generic fallback.997                bool src0_first_dim_contiguous = (op->src[0]->nb[0] == ggml_type_size(op->src[0]->type));998                bool src1_first_dim_contiguous = (op->src[1]->nb[0] == ggml_type_size(op->src[1]->type));999                bool dst_first_dim_contiguous  = (op->nb[0] == sizeof(float));1000 1001                bool dst_properly_ordered = true;1002                for (int d = 0; d < 3; d++) {1003                    if (op->ne[d] > 1 && op->ne[d + 1] > 1 && op->nb[d] > op->nb[d + 1]) {1004                        dst_properly_ordered = false;1005                    }1006                }1007 1008                supported = src0_first_dim_contiguous && src1_first_dim_contiguous && dst_first_dim_contiguous &&1009                            dst_properly_ordered;1010            } else {1011                supported = false;1012            }1013            break;1014        case GGML_OP_MUL_MAT_ID:1015            // Support MUL_MAT_ID for Mixture of Experts: (Q8_0/Q4_0/F16/F32) x F32 -> F32 with I32 expert indices1016            // src0 (as): [K, M, n_expert] - expert weight matrices (can be quantized)1017            // src1 (b):  [K, n_expert_used, batch] - activations (F32)1018            // src2 (ids): [n_expert_used, batch] - expert selection indices (I32)1019            // dst: [M, n_expert_used, batch, 1] - output (F32)1020            if (op->type == GGML_TYPE_F32 && op->src[0] &&1021                (op->src[0]->type == GGML_TYPE_Q8_0 || op->src[0]->type == GGML_TYPE_Q4_0 ||1022                 op->src[0]->type == GGML_TYPE_F16 || op->src[0]->type == GGML_TYPE_F32) &&1023                op->src[1] && op->src[1]->type == GGML_TYPE_F32 && op->src[2] && op->src[2]->type == GGML_TYPE_I32) {1024                // Check first dimension contiguity requirements (matching CPU backend)1025                bool src0_first_dim_contiguous = (op->src[0]->nb[0] == ggml_type_size(op->src[0]->type));1026                bool src1_first_dim_contiguous = (op->src[1]->nb[0] == ggml_type_size(op->src[1]->type));1027                bool src2_first_dim_contiguous = (op->src[2]->nb[0] == ggml_type_size(op->src[2]->type));1028                bool dst_first_dim_contiguous  = (op->nb[0] == sizeof(float));1029 1030                // Check destination stride ordering (only for dimensions with ne > 1)1031                bool dst_properly_ordered = true;1032                for (int d = 0; d < 3; d++) {1033                    if (op->ne[d] > 1 && op->ne[d + 1] > 1 && op->nb[d] > op->nb[d + 1]) {1034                        dst_properly_ordered = false;1035                    }1036                }1037 1038                // Validate tensor dimension constraints from GGML definition1039                bool dims_valid = (op->src[0]->ne[3] == 1) &&  // as is 3d (one matrix per expert)1040                                  (op->src[1]->ne[3] == 1) &&  // b is 3d1041                                  (op->src[2]->ne[2] == 1 && op->src[2]->ne[3] == 1) &&  // ids is 2d1042                                  (op->src[2]->ne[1] == op->src[1]->ne[2]) &&    // must have expert list per b row1043                                  (op->src[0]->ne[0] == op->src[1]->ne[0]) &&    // K dimension must match1044                                  (op->src[2]->ne[0] % op->src[1]->ne[1] == 0);  // can broadcast1045 1046                supported = src0_first_dim_contiguous && src1_first_dim_contiguous && src2_first_dim_contiguous &&1047                            dst_first_dim_contiguous && dst_properly_ordered && dims_valid;1048            } else {1049                supported = false;1050            }1051            break;1052        case GGML_OP_ROPE:1053            // Support F32 x I32 -> F32 RoPE for the modes implemented by rope_f32.1054            if (op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 && op->src[1] &&1055                op->src[1]->type == GGML_TYPE_I32 && ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0])) {1056                const int  mode             = ggml_get_op_params_i32(op, 2);1057                const int  ndims            = ggml_get_op_params_i32(op, 1);1058                const bool is_normal        = mode == GGML_ROPE_TYPE_NORMAL;1059                const bool is_neox          = mode == GGML_ROPE_TYPE_NEOX;1060                const bool is_imrope        = mode == GGML_ROPE_TYPE_IMROPE;1061                const bool zero_view_offset = op->src[0]->view_src == nullptr || op->src[0]->view_offs == 0;1062                const bool has_sections = ggml_get_op_params_i32(op, 11) > 0 || ggml_get_op_params_i32(op, 12) > 0 ||1063                                          ggml_get_op_params_i32(op, 13) > 0;1064 1065                supported =1066                    zero_view_offset && ndims <= 512 &&1067                    (is_normal || (is_neox && ndims % 16 == 0) || (is_imrope && ndims % 16 == 0 && has_sections));1068            } else {1069                supported = false;1070            }1071            break;1072        case GGML_OP_RMS_NORM:1073            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1074                        op->ne[0] % 16 == 0 && ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0]);1075            break;1076        case GGML_OP_NORM:1077            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1078                        op->ne[0] % 16 == 0 && ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0]);1079            break;1080        case GGML_OP_L2_NORM:1081            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1082                        op->ne[0] % 16 == 0 && ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0]);1083            break;1084        case GGML_OP_GROUP_NORM:1085            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1086                        ggml_is_contiguous(op) && et_ggml_is_row_contiguous(op->src[0]) &&1087                        ggml_get_op_params_i32(op, 0) > 0;1088            break;1089        case GGML_OP_IM2COL:1090            supported = op->src[0] && op->src[1] &&1091                        ((op->type == GGML_TYPE_F32 && op->src[1]->type == GGML_TYPE_F32) ||1092                         (op->type == GGML_TYPE_F16 &&1093                          (op->src[1]->type == GGML_TYPE_F16 || op->src[1]->type == GGML_TYPE_F32))) &&1094                        ggml_is_contiguous(op) && ggml_is_contiguous(op->src[1]) &&1095                        op->nb[0] == ggml_type_size(op->type) && op->src[1]->nb[0] == ggml_type_size(op->src[1]->type);1096            break;1097        case GGML_OP_CONV_2D:1098            {1099                // First-cut conv_2d_f32_me kernel constraints. Anything outside1100                // this falls back to CPU (it's a strict subset on purpose).1101                if (!op->src[0] || !op->src[1]) {1102                    supported = false;1103                    break;1104                }1105                if (op->type != GGML_TYPE_F32 || op->src[0]->type != GGML_TYPE_F32 ||1106                    op->src[1]->type != GGML_TYPE_F32) {1107                    supported = false;1108                    break;1109                }1110                if (!ggml_is_contiguous(op) || !ggml_is_contiguous(op->src[0]) || !ggml_is_contiguous(op->src[1])) {1111                    supported = false;1112                    break;1113                }1114 1115                const ggml_tensor * flt = op->src[0];  // [Kw, Kh, Cin, Cout]1116                const ggml_tensor * in  = op->src[1];  // [W,  H,  Cin, N]1117                const int32_t       s0  = ggml_get_op_params_i32(op, 0);1118                const int32_t       s1  = ggml_get_op_params_i32(op, 1);1119                const int32_t       p0  = ggml_get_op_params_i32(op, 2);1120                const int32_t       p1  = ggml_get_op_params_i32(op, 3);1121                const int32_t       d0  = ggml_get_op_params_i32(op, 4);1122                const int32_t       d1  = ggml_get_op_params_i32(op, 5);1123 1124                const int64_t Kw   = flt->ne[0];1125                const int64_t Kh   = flt->ne[1];1126                const int64_t Cin  = flt->ne[2];1127                const int64_t Cout = flt->ne[3];1128                const int64_t H    = in->ne[1];1129                (void) in->ne[0];1130 1131                if (s0 < 1 || s1 < 1 || !(d0 == 1 && d1 == 1) || Cin % 16 != 0 || Cout % 16 != 0 || in->ne[3] != 1) {1132                    supported = false;1133                    break;1134                }1135                const int64_t OW = op->ne[0];1136                const int64_t OH = op->ne[1];1137                if (OW <= 0 || OH <= 0) {1138                    supported = false;1139                    break;1140                }1141                (void) p0;1142                (void) p1;1143 1144                // Mirror the kernel's sizing:1145                //   if K_TILES * per_KT_bytes <= budget: 1 buffer, n_chunks=11146                //   else: 2 buffers (double-buffer), shrink chunk_KT until1147                //         2*chunk_KT*per_KT_bytes <= budget.1148                const int64_t Hp            = H + 2 * p1;1149                const int64_t OW_pad        = (OW + 15) & ~15;1150                const int64_t Wp_a          = OW_pad;1151                const bool    need_stage    = (OW % 16 != 0);1152                const int64_t stage_bytes   = need_stage ? (Cout * OH * OW_pad * 4) : 0;1153                const int64_t L2SCP_BUDGET  = 1500 * 1024;1154                // Per-hart partial-TenC scratch (mirrors kernel MAX_TILES_PER_HART=2):1155                // 32 minions x 2 tiles x 1024 bytes = 64 KB per shire.1156                const int64_t scratch_bytes = 32 * 2 * 16 * 16 * 4;1157                const int64_t budget        = L2SCP_BUDGET - stage_bytes - scratch_bytes;1158                const int64_t per_KT_bytes  = Kh * Kw * Cout * 16 * 4 + Kw * 16 * Hp * Wp_a * 4;1159                const int64_t K_TILES       = Cin / 16;1160 1161                int64_t chunk_KT_calc;1162                int64_t n_chunks_calc;1163                if (K_TILES * per_KT_bytes <= budget) {1164                    chunk_KT_calc = K_TILES;1165                    n_chunks_calc = 1;1166                } else {1167                    chunk_KT_calc = K_TILES;1168                    while (chunk_KT_calc > 1 && 2 * chunk_KT_calc * per_KT_bytes > budget) {1169                        chunk_KT_calc--;1170                    }1171                    while (chunk_KT_calc > 1 && K_TILES % chunk_KT_calc != 0) {1172                        chunk_KT_calc--;1173                    }1174                    if (chunk_KT_calc < 1) {1175                        supported = false;1176                        break;1177                    }1178                    n_chunks_calc = K_TILES / chunk_KT_calc;1179                }1180 1181                if (n_chunks_calc > 1) {1182                    const int64_t M_TILES     = Cout / 16;1183                    const int64_t w_tiles     = (OW + 15) / 16;1184                    const int64_t total_tiles = OH * w_tiles * M_TILES;1185                    // MAX_TILES_PER_HART = 2 (mirrors kernel constant).1186                    const int64_t max_workers = (need_stage ? 32 : 1024) * 2;1187                    if (total_tiles > max_workers) {1188                        supported = false;1189                        break;1190                    }1191                }1192 1193                supported = true;1194                break;1195            }1196        case GGML_OP_SCALE:1197            // F32 contiguous, total elements must be cache line aligned (16 floats)1198            supported = op->type == GGML_TYPE_F32 && op->src[0] && op->src[0]->type == GGML_TYPE_F32 &&1199                        ggml_is_contiguous(op) && ggml_is_contiguous(op->src[0]) && (ggml_nelements(op) % 16 == 0);1200            break;

Showing the first 1,200 of 1877 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai