Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
ggml-backend.cpp2003 linesDownload Raw Back to src
1// Note: porting this file to C++ is a work in progress2 3#ifdef _WIN324#define WIN32_LEAN_AND_MEAN5#ifndef NOMINMAX6#   define NOMINMAX7#endif8#include <windows.h>9#endif10 11#include "ggml-backend.h"12#include "ggml-backend-impl.h"13#include "ggml-alloc.h"14#include "ggml-impl.h"15 16#include <assert.h>17#include <limits.h>18#include <stdarg.h>19#include <stdio.h>20#include <stdlib.h>21#include <string.h>22#include <string>23#include <vector>24 25#ifdef __APPLE__26#include <sys/types.h>27#include <sys/sysctl.h>28#endif29 30 31// backend buffer type32 33const char * ggml_backend_buft_name(ggml_backend_buffer_type_t buft) {34    return buft->iface.get_name(buft);35}36 37ggml_backend_buffer_t ggml_backend_buft_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) {38    if (size == 0) {39        // return a dummy buffer for zero-sized allocations40        return ggml_backend_buffer_init(buft, {}, NULL, 0);41    }42 43    return buft->iface.alloc_buffer(buft, size);44}45 46size_t ggml_backend_buft_get_alignment(ggml_backend_buffer_type_t buft) {47    return buft->iface.get_alignment(buft);48}49 50size_t ggml_backend_buft_get_max_size(ggml_backend_buffer_type_t buft) {51    // get_max_size is optional, defaults to SIZE_MAX52    if (buft->iface.get_max_size) {53        return buft->iface.get_max_size(buft);54    }55    return SIZE_MAX;56}57 58size_t ggml_backend_buft_get_alloc_size(ggml_backend_buffer_type_t buft, struct ggml_tensor * tensor) {59    // get_alloc_size is optional, defaults to ggml_nbytes60    if (buft->iface.get_alloc_size) {61        size_t size = buft->iface.get_alloc_size(buft, tensor);62        assert(size >= ggml_nbytes(tensor));63        return size;64    }65    return ggml_nbytes(tensor);66}67 68bool ggml_backend_buft_is_host(ggml_backend_buffer_type_t buft) {69    if (buft->iface.is_host) {70        return buft->iface.is_host(buft);71    }72    return false;73}74 75ggml_backend_dev_t ggml_backend_buft_get_device(ggml_backend_buffer_type_t buft) {76    return buft->device;77}78 79// backend buffer80 81ggml_backend_buffer_t ggml_backend_buffer_init(82               ggml_backend_buffer_type_t buft,83        struct ggml_backend_buffer_i      iface,84               void *                     context,85               size_t                     size) {86    ggml_backend_buffer_t buffer = new ggml_backend_buffer {87        /* .interface = */ iface,88        /* .buft      = */ buft,89        /* .context   = */ context,90        /* .size      = */ size,91        /* .usage     = */ GGML_BACKEND_BUFFER_USAGE_ANY92    };93 94    return buffer;95}96 97const char * ggml_backend_buffer_name(ggml_backend_buffer_t buffer) {98    return ggml_backend_buft_name(ggml_backend_buffer_get_type(buffer));99}100 101void ggml_backend_buffer_free(ggml_backend_buffer_t buffer) {102    if (buffer == NULL) {103        return;104    }105 106    if (buffer->iface.free_buffer != NULL) {107        buffer->iface.free_buffer(buffer);108    }109    delete buffer;110}111 112size_t ggml_backend_buffer_get_size(ggml_backend_buffer_t buffer) {113    return buffer->size;114}115 116void * ggml_backend_buffer_get_base(ggml_backend_buffer_t buffer) {117    // get_base is optional if the buffer is zero-sized118    if (buffer->size == 0) {119        return NULL;120    }121 122    void * base = buffer->iface.get_base(buffer);123 124    GGML_ASSERT(base != NULL && "backend buffer base cannot be NULL");125 126    return base;127}128 129void ggml_backend_buffer_init_tensor(ggml_backend_buffer_t buffer, struct ggml_tensor * tensor) {130    // init_tensor is optional131    if (buffer->iface.init_tensor) {132        buffer->iface.init_tensor(buffer, tensor);133    }134}135 136void ggml_backend_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) {137    // clear is optional if the buffer is zero-sized138    if (buffer->size == 0) {139        return;140    }141 142    buffer->iface.clear(buffer, value);143}144 145size_t ggml_backend_buffer_get_alignment(ggml_backend_buffer_t buffer) {146    return ggml_backend_buft_get_alignment(ggml_backend_buffer_get_type(buffer));147}148 149size_t ggml_backend_buffer_get_max_size(ggml_backend_buffer_t buffer) {150    return ggml_backend_buft_get_max_size(ggml_backend_buffer_get_type(buffer));151}152 153size_t ggml_backend_buffer_get_alloc_size(ggml_backend_buffer_t buffer, struct ggml_tensor * tensor) {154    return ggml_backend_buft_get_alloc_size(ggml_backend_buffer_get_type(buffer), tensor);155}156 157bool ggml_backend_buffer_is_host(ggml_backend_buffer_t buffer) {158    return ggml_backend_buft_is_host(ggml_backend_buffer_get_type(buffer));159}160 161void ggml_backend_buffer_set_usage(ggml_backend_buffer_t buffer, enum ggml_backend_buffer_usage usage) {162    buffer->usage = usage;163 164    // FIXME: add a generic callback to the buffer interface165    if (ggml_backend_buffer_is_multi_buffer(buffer)) {166        ggml_backend_multi_buffer_set_usage(buffer, usage);167    }168}169 170enum ggml_backend_buffer_usage ggml_backend_buffer_get_usage(ggml_backend_buffer_t buffer) {171    return buffer->usage;172}173 174ggml_backend_buffer_type_t ggml_backend_buffer_get_type(ggml_backend_buffer_t buffer) {175    return buffer->buft;176}177 178void ggml_backend_buffer_reset(ggml_backend_buffer_t buffer) {179    if (buffer->iface.reset) {180        buffer->iface.reset(buffer);181    }182}183 184bool ggml_backend_buffer_copy_tensor(const struct ggml_tensor * src, struct ggml_tensor * dst) {185    ggml_backend_buffer_t dst_buf = dst->view_src ? dst->view_src->buffer : dst->buffer;186    if (dst_buf->iface.cpy_tensor) {187        return dst_buf->iface.cpy_tensor(dst_buf, src, dst);188    }189    return false;190}191 192// backend193 194ggml_guid_t ggml_backend_guid(ggml_backend_t backend) {195    if (backend == NULL) {196        return NULL;197    }198    return backend->guid;199}200 201const char * ggml_backend_name(ggml_backend_t backend) {202    if (backend == NULL) {203        return "NULL";204    }205    return backend->iface.get_name(backend);206}207 208void ggml_backend_free(ggml_backend_t backend) {209    if (backend == NULL) {210        return;211    }212 213    backend->iface.free(backend);214}215 216ggml_backend_buffer_type_t ggml_backend_get_default_buffer_type(ggml_backend_t backend) {217    return ggml_backend_dev_buffer_type(backend->device);218}219 220ggml_backend_buffer_t ggml_backend_alloc_buffer(ggml_backend_t backend, size_t size) {221    return ggml_backend_buft_alloc_buffer(ggml_backend_get_default_buffer_type(backend), size);222}223 224size_t ggml_backend_get_alignment(ggml_backend_t backend) {225    return ggml_backend_buft_get_alignment(ggml_backend_get_default_buffer_type(backend));226}227 228size_t ggml_backend_get_max_size(ggml_backend_t backend) {229    return ggml_backend_buft_get_max_size(ggml_backend_get_default_buffer_type(backend));230}231 232void ggml_backend_tensor_set_async(ggml_backend_t backend, struct ggml_tensor * tensor, const void * data, size_t offset, size_t size) {233    GGML_ASSERT(tensor->data != NULL && "tensor not allocated");234    GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor write out of bounds");235 236    if (backend->iface.set_tensor_async == NULL) {237        ggml_backend_tensor_set(tensor, data, offset, size);238    } else {239        backend->iface.set_tensor_async(backend, tensor, data, offset, size);240    }241}242 243void ggml_backend_tensor_get_async(ggml_backend_t backend, const struct ggml_tensor * tensor, void * data, size_t offset, size_t size) {244    GGML_ASSERT(tensor->data != NULL && "tensor not allocated");245    GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor read out of bounds");246 247    if (backend->iface.get_tensor_async == NULL) {248        ggml_backend_tensor_get(tensor, data, offset, size);249    } else {250        backend->iface.get_tensor_async(backend, tensor, data, offset, size);251    }252}253 254void ggml_backend_tensor_set(struct ggml_tensor * tensor, const void * data, size_t offset, size_t size) {255    GGML_ASSERT(tensor);256    ggml_backend_buffer_t buf = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;257 258    if (size == 0) {259        return;260    }261 262    GGML_ASSERT(buf != NULL && "tensor buffer not set");263    GGML_ASSERT(tensor->data != NULL && "tensor not allocated");264    GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor write out of bounds");265 266    buf->iface.set_tensor(buf, tensor, data, offset, size);267}268 269void ggml_backend_tensor_get(const struct ggml_tensor * tensor, void * data, size_t offset, size_t size) {270    GGML_ASSERT(tensor);271    ggml_backend_buffer_t buf = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;272 273    if (size == 0) {274        return;275    }276 277    GGML_ASSERT(buf != NULL && "tensor buffer not set");278    GGML_ASSERT(tensor->data != NULL && "tensor not allocated");279    GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor read out of bounds");280 281    buf->iface.get_tensor(buf, tensor, data, offset, size);282}283 284void ggml_backend_tensor_memset(struct ggml_tensor * tensor, uint8_t value, size_t offset, size_t size) {285    ggml_backend_buffer_t buf = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;286 287    if (size == 0) {288        return;289    }290 291    GGML_ASSERT(buf != NULL && "tensor buffer not set");292    GGML_ASSERT(tensor->data != NULL && "tensor not allocated");293    GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor write out of bounds");294    GGML_ASSERT(buf->iface.memset_tensor != NULL && "memset not implemented by backend buffer");295 296    buf->iface.memset_tensor(buf, tensor, value, offset, size);297}298 299void ggml_backend_synchronize(ggml_backend_t backend) {300    if (backend->iface.synchronize == NULL) {301        return;302    }303 304    backend->iface.synchronize(backend);305}306 307ggml_backend_graph_plan_t ggml_backend_graph_plan_create(ggml_backend_t backend, struct ggml_cgraph * cgraph) {308    GGML_ASSERT(backend->iface.graph_plan_create != NULL);309 310    return backend->iface.graph_plan_create(backend, cgraph);311}312 313void ggml_backend_graph_plan_free(ggml_backend_t backend, ggml_backend_graph_plan_t plan) {314    GGML_ASSERT(backend->iface.graph_plan_free != NULL);315 316    backend->iface.graph_plan_free(backend, plan);317}318 319enum ggml_status ggml_backend_graph_plan_compute(ggml_backend_t backend, ggml_backend_graph_plan_t plan) {320    GGML_ASSERT(backend->iface.graph_plan_compute != NULL);321 322    return backend->iface.graph_plan_compute(backend, plan);323}324 325enum ggml_status ggml_backend_graph_compute(ggml_backend_t backend, struct ggml_cgraph * cgraph) {326    enum ggml_status err = ggml_backend_graph_compute_async(backend, cgraph);327    ggml_backend_synchronize(backend);328    return err;329}330 331enum ggml_status ggml_backend_graph_compute_async(ggml_backend_t backend, struct ggml_cgraph * cgraph) {332    return backend->iface.graph_compute(backend, cgraph);333}334 335bool ggml_backend_supports_op(ggml_backend_t backend, const struct ggml_tensor * op) {336    return ggml_backend_dev_supports_op(backend->device, op);337}338 339bool ggml_backend_supports_buft(ggml_backend_t backend, ggml_backend_buffer_type_t buft) {340    return ggml_backend_dev_supports_buft(backend->device, buft);341}342 343bool ggml_backend_offload_op(ggml_backend_t backend, const struct ggml_tensor * op) {344    return ggml_backend_dev_offload_op(backend->device, op);345}346 347ggml_backend_dev_t ggml_backend_get_device(ggml_backend_t backend) {348    return backend->device;349}350 351// backend copy352 353static bool ggml_are_same_layout(const struct ggml_tensor * a, const struct ggml_tensor * b) {354    if (a->type != b->type) {355        return false;356    }357    for (int i = 0; i < GGML_MAX_DIMS; i++) {358        if (a->ne[i] != b->ne[i]) {359            return false;360        }361        if (a->nb[i] != b->nb[i]) {362            return false;363        }364    }365    return true;366}367 368void ggml_backend_tensor_copy(struct ggml_tensor * src, struct ggml_tensor * dst) {369    GGML_ASSERT(ggml_are_same_layout(src, dst) && "cannot copy tensors with different layouts");370 371    if (src == dst) {372        return;373    }374 375    if (ggml_backend_buffer_is_host(src->buffer)) {376        ggml_backend_tensor_set(dst, src->data, 0, ggml_nbytes(src));377    } else if (ggml_backend_buffer_is_host(dst->buffer)) {378        ggml_backend_tensor_get(src, dst->data, 0, ggml_nbytes(src));379    } else if (!ggml_backend_buffer_copy_tensor(src, dst)) {380#ifndef NDEBUG381        GGML_LOG_DEBUG("%s: warning: slow copy from %s to %s\n", __func__, ggml_backend_buffer_name(src->buffer), ggml_backend_buffer_name(dst->buffer));382#endif383        size_t nbytes = ggml_nbytes(src);384        void * data = malloc(nbytes);385        ggml_backend_tensor_get(src, data, 0, nbytes);386        ggml_backend_tensor_set(dst, data, 0, nbytes);387        free(data);388    }389}390 391void ggml_backend_tensor_copy_async(ggml_backend_t backend_src, ggml_backend_t backend_dst, struct ggml_tensor * src, struct ggml_tensor * dst) {392    GGML_ASSERT(ggml_are_same_layout(src, dst) && "cannot copy tensors with different layouts");393 394    if (src == dst) {395        return;396    }397 398    if (backend_dst->iface.cpy_tensor_async != NULL) {399        if (backend_dst->iface.cpy_tensor_async(backend_src, backend_dst, src, dst)) {400            return;401        }402    }403 404    // an async copy would normally happen after all the queued operations on both backends are completed405    // to simulate the same behavior, we need to synchronize both backends first, and do a blocking copy406    ggml_backend_synchronize(backend_src);407    ggml_backend_synchronize(backend_dst);408    ggml_backend_tensor_copy(src, dst);409}410 411// events412 413ggml_backend_event_t ggml_backend_event_new(ggml_backend_dev_t device) {414    // null device is allowed for the transition period to the device interface415    if (device == NULL || device->iface.event_new == NULL) {416        return NULL;417    }418    return device->iface.event_new(device);419}420 421void ggml_backend_event_free(ggml_backend_event_t event) {422    if (event == NULL) {423        return;424    }425    event->device->iface.event_free(event->device, event);426}427 428void ggml_backend_event_record(ggml_backend_event_t event, ggml_backend_t backend) {429    GGML_ASSERT(backend->iface.event_record != NULL);430 431    backend->iface.event_record(backend, event);432}433 434void ggml_backend_event_synchronize(ggml_backend_event_t event) {435    GGML_ASSERT(event->device->iface.event_synchronize);436 437    event->device->iface.event_synchronize(event->device, event);438}439 440void ggml_backend_event_wait(ggml_backend_t backend, ggml_backend_event_t event) {441    GGML_ASSERT(backend->iface.event_wait != NULL);442 443    backend->iface.event_wait(backend, event);444}445 446// Backend device447 448const char * ggml_backend_dev_name(ggml_backend_dev_t device) {449    return device->iface.get_name(device);450}451 452const char * ggml_backend_dev_description(ggml_backend_dev_t device) {453    return device->iface.get_description(device);454}455 456void ggml_backend_dev_memory(ggml_backend_dev_t device, size_t * free, size_t * total) {457    device->iface.get_memory(device, free, total);458}459 460enum ggml_backend_dev_type ggml_backend_dev_type(ggml_backend_dev_t device) {461    return device->iface.get_type(device);462}463 464void ggml_backend_dev_get_props(ggml_backend_dev_t device, struct ggml_backend_dev_props * props) {465    memset(props, 0, sizeof(*props));466    device->iface.get_props(device, props);467}468 469ggml_backend_reg_t ggml_backend_dev_backend_reg(ggml_backend_dev_t device) {470    return device->reg;471}472 473ggml_backend_t ggml_backend_dev_init(ggml_backend_dev_t device, const char * params) {474    return device->iface.init_backend(device, params);475}476 477ggml_backend_buffer_type_t ggml_backend_dev_buffer_type(ggml_backend_dev_t device) {478    return device->iface.get_buffer_type(device);479}480 481ggml_backend_buffer_type_t ggml_backend_dev_host_buffer_type(ggml_backend_dev_t device) {482    if (device->iface.get_host_buffer_type == NULL) {483        return NULL;484    }485 486    return device->iface.get_host_buffer_type(device);487}488 489ggml_backend_buffer_t ggml_backend_dev_buffer_from_host_ptr(ggml_backend_dev_t device, void * ptr, size_t size, size_t max_tensor_size) {490    return device->iface.buffer_from_host_ptr(device, ptr, size, max_tensor_size);491}492 493bool ggml_backend_dev_supports_op(ggml_backend_dev_t device, const struct ggml_tensor * op) {494    return device->iface.supports_op(device, op);495}496 497bool ggml_backend_dev_supports_buft(ggml_backend_dev_t device, ggml_backend_buffer_type_t buft) {498    return device->iface.supports_buft(device, buft);499}500 501bool ggml_backend_dev_offload_op(ggml_backend_dev_t device, const struct ggml_tensor * op) {502    if (device->iface.offload_op != NULL) {503        return device->iface.offload_op(device, op);504    }505 506    return false;507}508 509// Backend (reg)510 511const char * ggml_backend_reg_name(ggml_backend_reg_t reg) {512    return reg->iface.get_name(reg);513}514 515size_t ggml_backend_reg_dev_count(ggml_backend_reg_t reg) {516    return reg->iface.get_device_count(reg);517}518 519ggml_backend_dev_t ggml_backend_reg_dev_get(ggml_backend_reg_t reg, size_t index) {520    return reg->iface.get_device(reg, index);521}522 523void * ggml_backend_reg_get_proc_address(ggml_backend_reg_t reg, const char * name) {524    if (!reg->iface.get_proc_address) {525        return NULL;526    }527    return reg->iface.get_proc_address(reg, name);528}529 530// multi-buffer buffer531 532struct ggml_backend_multi_buffer_context {533    ggml_backend_buffer_t * buffers;534    size_t n_buffers;535};536 537static void ggml_backend_multi_buffer_free_buffer(ggml_backend_buffer_t buffer) {538    ggml_backend_multi_buffer_context * ctx = (ggml_backend_multi_buffer_context *) buffer->context;539    for (size_t i = 0; i < ctx->n_buffers; i++) {540        ggml_backend_buffer_free(ctx->buffers[i]);541    }542 543    free(ctx->buffers);544    free(ctx);545}546 547static void ggml_backend_multi_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) {548    ggml_backend_multi_buffer_context * ctx = (ggml_backend_multi_buffer_context *) buffer->context;549    for (size_t i = 0; i < ctx->n_buffers; i++) {550        ggml_backend_buffer_clear(ctx->buffers[i], value);551    }552}553 554static const struct ggml_backend_buffer_i ggml_backend_multi_buffer_i = {555    /* .free_buffer     = */ ggml_backend_multi_buffer_free_buffer,556    /* .get_base        = */ NULL,557    /* .init_tensor     = */ NULL,558    /* .memset_tensor   = */ NULL,559    /* .set_tensor      = */ NULL,560    /* .get_tensor      = */ NULL,561    /* .cpy_tensor      = */ NULL,562    /* .clear           = */ ggml_backend_multi_buffer_clear,563    /* .reset           = */ NULL,564};565 566ggml_backend_buffer_t ggml_backend_multi_buffer_alloc_buffer(ggml_backend_buffer_t * buffers, size_t n_buffers) {567    ggml_backend_multi_buffer_context * ctx = (ggml_backend_multi_buffer_context *) malloc(sizeof(struct ggml_backend_multi_buffer_context));568    ctx->n_buffers = n_buffers;569    ctx->buffers = (ggml_backend_buffer_t *) malloc(n_buffers * sizeof(ggml_backend_buffer_t));570 571    GGML_ASSERT(ctx->buffers != NULL);572 573    size_t total_size = 0;574    for (size_t i = 0; i < n_buffers; i++) {575        ctx->buffers[i] = buffers[i];576        total_size += ggml_backend_buffer_get_size(buffers[i]);577    }578 579    return ggml_backend_buffer_init(buffers[0]->buft, ggml_backend_multi_buffer_i, ctx, total_size);580}581 582bool ggml_backend_buffer_is_multi_buffer(ggml_backend_buffer_t buffer) {583    return buffer->iface.free_buffer == ggml_backend_multi_buffer_free_buffer;584}585 586void ggml_backend_multi_buffer_set_usage(ggml_backend_buffer_t buffer, enum ggml_backend_buffer_usage usage) {587    GGML_ASSERT(ggml_backend_buffer_is_multi_buffer(buffer));588    ggml_backend_multi_buffer_context * ctx = (ggml_backend_multi_buffer_context *) buffer->context;589    for (size_t i = 0; i < ctx->n_buffers; i++) {590        ggml_backend_buffer_set_usage(ctx->buffers[i], usage);591    }592}593 594// creates a copy of the tensor with the same memory layout595static struct ggml_tensor * ggml_dup_tensor_layout(struct ggml_context * ctx, const struct ggml_tensor * tensor) {596    struct ggml_tensor * dup = ggml_dup_tensor(ctx, tensor);597    for (int i = 0; i < GGML_MAX_DIMS; i++) {598        dup->nb[i] = tensor->nb[i];599    }600    return dup;601}602 603static bool ggml_is_view_op(enum ggml_op op) {604    return op == GGML_OP_VIEW || op == GGML_OP_RESHAPE || op == GGML_OP_PERMUTE || op == GGML_OP_TRANSPOSE;605}606 607// scheduler608 609#ifndef GGML_SCHED_MAX_BACKENDS610#define GGML_SCHED_MAX_BACKENDS 16611#endif612 613#ifndef GGML_SCHED_MAX_SPLIT_INPUTS614#define GGML_SCHED_MAX_SPLIT_INPUTS GGML_MAX_SRC615#endif616 617#ifndef GGML_SCHED_MAX_COPIES618#define GGML_SCHED_MAX_COPIES 4619#endif620 621struct ggml_backend_sched_split {622    int backend_id;623    int i_start;624    int i_end;625    struct ggml_tensor * inputs[GGML_SCHED_MAX_SPLIT_INPUTS];626    int n_inputs;627    // graph view of this split628    struct ggml_cgraph graph;629};630 631struct ggml_backend_sched {632    bool is_reset; // true if the scheduler has been reset since the last graph split633    bool is_alloc;634 635    int n_backends;636 637    ggml_backend_t backends[GGML_SCHED_MAX_BACKENDS];638    ggml_backend_buffer_type_t bufts[GGML_SCHED_MAX_BACKENDS];639    ggml_gallocr_t galloc;640 641    // hash map of the nodes in the graph642    struct ggml_hash_set  hash_set;643    int                 * hv_tensor_backend_ids; // [hash_set.size]644    struct ggml_tensor ** hv_tensor_copies;      // [hash_set.size][n_backends][n_copies]645 646    int * node_backend_ids; // [graph_size]647    int * leaf_backend_ids; // [graph_size]648 649    int * prev_node_backend_ids; // [graph_size]650    int * prev_leaf_backend_ids; // [graph_size]651 652    // copy of the graph with modified inputs653    struct ggml_cgraph graph;654 655    // graph splits656    struct ggml_backend_sched_split * splits;657    int n_splits;658    int splits_capacity;659 660    // pipeline parallelism support661    int n_copies;662    int cur_copy;663    ggml_backend_event_t events[GGML_SCHED_MAX_BACKENDS][GGML_SCHED_MAX_COPIES];664    struct ggml_tensor * graph_inputs[GGML_SCHED_MAX_SPLIT_INPUTS];665    int n_graph_inputs;666 667    struct ggml_context * ctx;668 669    ggml_backend_sched_eval_callback callback_eval;670    void * callback_eval_user_data;671 672    char * context_buffer;673    size_t context_buffer_size;674 675    int debug;676};677 678#define hash_id(tensor) ggml_hash_find_or_insert(&sched->hash_set, tensor)679#define tensor_backend_id(tensor) sched->hv_tensor_backend_ids[hash_id(tensor)]680#define tensor_id_copy(id, backend_id, copy_id) sched->hv_tensor_copies[(id) * sched->n_backends * sched->n_copies + (backend_id) * sched->n_copies + (copy_id)]681#define tensor_copy(tensor, backend_id, copy_id) tensor_id_copy(hash_id(tensor), backend_id, copy_id)682 683// returns the priority of the backend, lower id is higher priority684static int ggml_backend_sched_backend_id(ggml_backend_sched_t sched, ggml_backend_t backend) {685    for (int i = 0; i < sched->n_backends; i++) {686        if (sched->backends[i] == backend) {687            return i;688        }689    }690    return -1;691}692 693static int ggml_backend_sched_backend_from_buffer(ggml_backend_sched_t sched, const struct ggml_tensor * tensor, const struct ggml_tensor * op) {694    ggml_backend_buffer_t buffer = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;695    if (buffer == NULL) {696        return -1;697    }698 699    // find highest prio backend that supports the buffer type and the op700    for (int i = 0; i < sched->n_backends; i++) {701        if (ggml_backend_supports_buft(sched->backends[i], buffer->buft) &&702            ggml_backend_supports_op(sched->backends[i], op)) {703            return i;704        }705    }706 707#ifndef NDEBUG708    GGML_LOG_DEBUG("%s: warning: no backend supports op %s with a weight with buffer type %s used in tensor %s, the weight will need to be copied\n",709        __func__, ggml_op_desc(tensor), ggml_backend_buffer_name(buffer), tensor->name);710#endif711 712    return -1;713}714 715#if 0716#define GGML_SCHED_MAX_SPLITS_DEBUG 4096717static char causes[GGML_DEFAULT_GRAPH_SIZE*16 + GGML_SCHED_MAX_SPLITS_DEBUG*GGML_SCHED_MAX_SPLIT_INPUTS][128]; // debug only718#define SET_CAUSE(node, ...) sprintf(causes[hash_id(node)], __VA_ARGS__)719#define GET_CAUSE(node) causes[hash_id(node)]720#else721#define SET_CAUSE(node, ...)722#define GET_CAUSE(node) ""723#endif724 725// returns the backend that should be used for the node based on the current locations726static int ggml_backend_sched_backend_id_from_cur(ggml_backend_sched_t sched, struct ggml_tensor * tensor) {727    // assign pre-allocated nodes to their backend728    int cur_backend_id = ggml_backend_sched_backend_from_buffer(sched, tensor, tensor);729    if (cur_backend_id != -1) {730        SET_CAUSE(tensor, "1.dst");731        return cur_backend_id;732    }733 734    // view_src735    if (tensor->view_src != NULL) {736        cur_backend_id = ggml_backend_sched_backend_from_buffer(sched, tensor->view_src, tensor);737        if (cur_backend_id != -1) {738            SET_CAUSE(tensor, "1.vsrc");739            return cur_backend_id;740        }741    }742 743    if (tensor->buffer || (tensor->view_src && tensor->view_src->buffer)) {744        // since the tensor is pre-allocated, it cannot be moved to another backend745        ggml_backend_buffer_t buffer = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;746        GGML_ABORT("pre-allocated tensor (%s) in a buffer (%s) that cannot run the operation (%s)", tensor->name, ggml_backend_buffer_name(buffer), ggml_op_name(tensor->op));747    }748 749    // graph input750    if (tensor->flags & GGML_TENSOR_FLAG_INPUT) {751        cur_backend_id = sched->n_backends - 1; // last backend (assumed CPU)752        SET_CAUSE(tensor, "1.inp");753        return cur_backend_id;754    }755 756    // operations with weights are preferably run on the same backend as the weights757    for (int i = 0; i < GGML_MAX_SRC; i++) {758        const struct ggml_tensor * src = tensor->src[i];759        if (src == NULL) {760            continue;761        }762        // skip ROPE since the rope freqs tensor is too small to choose a backend based on it763        // not an ideal solution764        if (tensor->op != GGML_OP_ROPE && src->buffer != NULL && src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {765            int src_backend_id = ggml_backend_sched_backend_from_buffer(sched, src, tensor);766            // check if a backend with higher prio wants to offload the op767            if (src_backend_id == sched->n_backends - 1 && ggml_backend_buffer_is_host(src->buffer)) {768                for (int b = 0; b < src_backend_id; b++) {769                    if (ggml_backend_supports_op(sched->backends[b], tensor) && ggml_backend_offload_op(sched->backends[b], tensor)) {770                        SET_CAUSE(tensor, "1.off");771                        return b;772                    }773                }774            }775            SET_CAUSE(tensor, "1.wgt%d", i);776            return src_backend_id;777        }778    }779 780    return -1;781}782 783static char * fmt_size(size_t size) {784    static char buffer[128];785    if (size >= 1024*1024) {786        snprintf(buffer, sizeof(buffer), "%zuM", size/1024/1024);787    } else {788        snprintf(buffer, sizeof(buffer), "%zuK", size/1024);789    }790    return buffer;791}792 793static void ggml_backend_sched_print_assignments(ggml_backend_sched_t sched, struct ggml_cgraph * graph) {794    int cur_split = 0;795    for (int i = 0; i < graph->n_nodes; i++) {796        if (cur_split < sched->n_splits && i == sched->splits[cur_split].i_start) {797            ggml_backend_t split_backend = sched->backends[sched->splits[cur_split].backend_id];798            GGML_LOG_DEBUG("\n## SPLIT #%d: %s # %d inputs", cur_split, ggml_backend_name(split_backend),799                sched->splits[cur_split].n_inputs);800            for (int j = 0; j < sched->splits[cur_split].n_inputs; j++) {801                if (j == 0) {802                    GGML_LOG_DEBUG(": ");803                }804                GGML_LOG_DEBUG("[%s (%5.5s)] ", sched->splits[cur_split].inputs[j]->name,805                    fmt_size(ggml_nbytes(sched->splits[cur_split].inputs[j])));806            }807            GGML_LOG_DEBUG("\n");808            cur_split++;809        }810        struct ggml_tensor * node = graph->nodes[i];811        if (ggml_is_view_op(node->op)) {812            continue;813        }814        if (sched->debug > 1) {815            ggml_backend_t tensor_backend = ggml_backend_sched_get_tensor_backend(sched, node);816            GGML_LOG_DEBUG("node #%3d (%10.10s): %20.20s (%5.5s) [%5.5s %8.8s]:", i, ggml_op_name(node->op), node->name,817                fmt_size(ggml_nbytes(node)), tensor_backend ? ggml_backend_name(tensor_backend) : "NULL", GET_CAUSE(node));818            for (int j = 0; j < GGML_MAX_SRC; j++) {819                struct ggml_tensor * src = node->src[j];820                if (src == NULL) {821                    continue;822                }823                ggml_backend_t src_backend = ggml_backend_sched_get_tensor_backend(sched, src);824                GGML_LOG_DEBUG(" %20.20s (%5.5s) [%5.5s %8.8s]", src->name,825                    fmt_size(ggml_nbytes(src)), src_backend ? ggml_backend_name(src_backend) : "NULL", GET_CAUSE(src));826            }827            GGML_LOG_DEBUG("\n");828        }829    }830}831 832static bool ggml_backend_sched_buffer_supported(ggml_backend_sched_t sched, struct ggml_tensor * t, int backend_id) {833    ggml_backend_buffer_t buf = t->view_src ? t->view_src->buffer : t->buffer;834    ggml_backend_buffer_type_t buft = NULL;835 836    if (buf) {837        // the tensor is already allocated838        buft = buf->buft;839    } else {840        // see if the tensor already has a backend assigned, and use the buffer type of that backend841        int tensor_backend_id = tensor_backend_id(t);842        if (tensor_backend_id == -1 && t->view_src) {843            tensor_backend_id = tensor_backend_id(t->view_src);844        }845        if (tensor_backend_id != -1) {846            buft = sched->bufts[tensor_backend_id];847        }848    }849 850    return buft != NULL && ggml_backend_supports_buft(sched->backends[backend_id], buft);851}852 853static void ggml_backend_sched_set_if_supported(ggml_backend_sched_t sched, struct ggml_tensor * node, int cur_backend_id, int * node_backend_id) {854    if (ggml_backend_supports_op(sched->backends[cur_backend_id], node)) {855        *node_backend_id = cur_backend_id;856        SET_CAUSE(node, "2.sup");857    }858}859 860// assigns backends to ops and splits the graph into subgraphs that can be computed on the same backend861static void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgraph * graph) {862    // reset splits863    sched->n_splits = 0;864    sched->n_graph_inputs = 0;865    sched->is_reset = false;866 867    struct ggml_init_params params = {868        /* .mem_size =   */ sched->context_buffer_size,869        /* .mem_buffer = */ sched->context_buffer,870        /* .no_alloc =   */ true871    };872 873    ggml_free(sched->ctx);874 875    sched->ctx = ggml_init(params);876    if (sched->ctx == NULL) {877        GGML_ABORT("%s: failed to initialize context\n", __func__);878    }879 880    // pass 1: assign backends to ops with pre-allocated inputs881    for (int i = 0; i < graph->n_leafs; i++) {882        struct ggml_tensor * leaf = graph->leafs[i];883        int * leaf_backend_id = &tensor_backend_id(leaf);884        // do not overwrite user assignments885        if (*leaf_backend_id == -1) {886            *leaf_backend_id = ggml_backend_sched_backend_id_from_cur(sched, leaf);887        }888    }889 890    for (int i = 0; i < graph->n_nodes; i++) {891        struct ggml_tensor * node = graph->nodes[i];892        int * node_backend_id = &tensor_backend_id(node);893        // do not overwrite user assignments894        if (*node_backend_id == -1) {895            *node_backend_id = ggml_backend_sched_backend_id_from_cur(sched, node);896 897#if 0898            // src899            if (node->op == GGML_OP_NONE) {900                continue;901            }902 903            for (int j = 0; j < GGML_MAX_SRC; j++) {904                struct ggml_tensor * src = node->src[j];905                if (src == NULL) {906                    continue;907                }908                int * src_backend_id = &tensor_backend_id(src);909                if (*src_backend_id == -1) {910                    *src_backend_id = ggml_backend_sched_backend_id_from_cur(sched, src);911                }912            }913#endif914        }915    }916 917    // pass 2: expand current backend assignments918    // assign the same backend to adjacent nodes919    // expand gpu backends (i.e. non last prio) up and down, ignoring cpu (the lowest priority backend)920    // thus, cpu will never be used unless weights are on cpu, or there are no gpu ops between cpu ops921    // ops unsupported by the backend being expanded will be left unassigned so that they can be assigned later when the locations of its inputs are known922    // expand gpu down923    {924        int cur_backend_id = -1;925        for (int i = 0; i < graph->n_nodes; i++) {926            struct ggml_tensor * node = graph->nodes[i];927            if (ggml_is_view_op(node->op)) {928                continue;929            }930            int * node_backend_id = &tensor_backend_id(node);931            if (*node_backend_id != -1) {932                if (*node_backend_id == sched->n_backends - 1) {933                    // skip cpu (lowest prio backend)934                    cur_backend_id = -1;935                } else {936                    cur_backend_id = *node_backend_id;937                }938            } else if (cur_backend_id != -1) {939                ggml_backend_sched_set_if_supported(sched, node, cur_backend_id, node_backend_id);940            }941        }942    }943    // expand gpu up944    {945        int cur_backend_id = -1;946        for (int i = graph->n_nodes - 1; i >= 0; i--) {947            struct ggml_tensor * node = graph->nodes[i];948            if (ggml_is_view_op(node->op)) {949                continue;950            }951            int * node_backend_id = &tensor_backend_id(node);952            if (*node_backend_id != -1) {953                if (*node_backend_id == sched->n_backends - 1) {954                    // skip cpu (lowest prio backend)955                    cur_backend_id = -1;956                } else {957                    cur_backend_id = *node_backend_id;958                }959            } else if (cur_backend_id != -1) {960                ggml_backend_sched_set_if_supported(sched, node, cur_backend_id, node_backend_id);961            }962        }963    }964    // expand rest down965    {966        int cur_backend_id = -1;967        for (int i = 0; i < graph->n_nodes; i++) {968            struct ggml_tensor * node = graph->nodes[i];969            if (ggml_is_view_op(node->op)) {970                continue;971            }972            int * node_backend_id = &tensor_backend_id(node);973            if (*node_backend_id != -1) {974                cur_backend_id = *node_backend_id;975            } else if (cur_backend_id != -1) {976                ggml_backend_sched_set_if_supported(sched, node, cur_backend_id, node_backend_id);977            }978        }979    }980    // expand rest up981    {982        int cur_backend_id = -1;983        for (int i = graph->n_nodes - 1; i >= 0; i--) {984            struct ggml_tensor * node = graph->nodes[i];985            if (ggml_is_view_op(node->op)) {986                continue;987            }988            int * node_backend_id = &tensor_backend_id(node);989            if (*node_backend_id != -1) {990                cur_backend_id = *node_backend_id;991            } else if (cur_backend_id != -1) {992                ggml_backend_sched_set_if_supported(sched, node, cur_backend_id, node_backend_id);993            }994        }995    }996 997    // pass 3: upgrade nodes to higher prio backends with compatible buffer types998    // if the tensor is already in the same buffer type (*) as another higher priority backend, we should move it there999    // however, we also need to verify that the sources are in compatible buffer types1000    // (*) the actual requirement is more relaxed, the buffer type of the backend should be supported by all the users of this tensor further down the graph1001    // however, this is slow to verify, so we have a more strict requirement that the buffer type is the same1002    // this is not uncommon since multiple backends can use host memory, with the same buffer type (eg. BLAS and CPU)1003    // additionally, set remaining unassigned nodes to the backend with the most supported inputs1004    // only nodes that could not be assigned during expansion due to the backend not supporting the op should be unassigned at this point1005    for (int i = 0; i < graph->n_nodes; i++) {1006        struct ggml_tensor * node = graph->nodes[i];1007        if (ggml_is_view_op(node->op)) {1008            continue;1009        }1010        int * node_backend_id = &tensor_backend_id(node);1011        if (*node_backend_id == -1) {1012            // unassigned node: find the backend with the most supported inputs1013            int n_supported_best = -1;1014            for (int b = 0; b < sched->n_backends; b++) {1015                if (ggml_backend_supports_op(sched->backends[b], node)) {1016                    int n_supported = 0;1017                    for (int j = 0; j < GGML_MAX_SRC; j++) {1018                        struct ggml_tensor * src = node->src[j];1019                        if (src == NULL) {1020                            continue;1021                        }1022                        if ((tensor_backend_id(src) != -1 || tensor_backend_id(src->view_src) != -1) && ggml_backend_sched_buffer_supported(sched, src, b)) {1023                            n_supported++;1024                        }1025                    }1026                    if (n_supported > n_supported_best) {1027                        n_supported_best = n_supported;1028                        *node_backend_id = b;1029                        SET_CAUSE(node, "3.best");1030                    }1031                }1032            }1033        } else {1034            // assigned node: upgrade to higher prio backend if possible1035            for (int b = 0; b < *node_backend_id; b++) {1036                if (sched->bufts[b] == sched->bufts[*node_backend_id] && ggml_backend_supports_op(sched->backends[b], node)) {1037                    bool supported = true;1038                    for (int j = 0; j < GGML_MAX_SRC; j++) {1039                        struct ggml_tensor * src = node->src[j];1040                        if (src == NULL) {1041                            continue;1042                        }1043                        if (!ggml_backend_sched_buffer_supported(sched, src, b)) {1044                            supported = false;1045                            break;1046                        }1047                    }1048                    if (supported) {1049                        *node_backend_id = b;1050                        SET_CAUSE(node, "3.upg");1051                        break;1052                    }1053                }1054            }1055        }1056    }1057 1058    // pass 4: assign backends to remaining src from dst and view_src1059    for (int i = 0; i < graph->n_nodes; i++) {1060        struct ggml_tensor * node = graph->nodes[i];1061        int * cur_backend_id = &tensor_backend_id(node);1062        if (node->view_src != NULL && *cur_backend_id == -1) {1063            *cur_backend_id = tensor_backend_id(node->view_src);1064            SET_CAUSE(node, "4.vsrc");1065        }1066        for (int j = 0; j < GGML_MAX_SRC; j++) {1067            struct ggml_tensor * src = node->src[j];1068            if (src == NULL) {1069                continue;1070            }1071            int * src_backend_id = &tensor_backend_id(src);1072            if (*src_backend_id == -1) {1073                if (src->view_src != NULL) {1074                    // views are always on the same backend as the source1075                    *src_backend_id = tensor_backend_id(src->view_src);1076                    SET_CAUSE(src, "4.vsrc");1077                } else {1078                    *src_backend_id = *cur_backend_id;1079                    SET_CAUSE(src, "4.cur");1080                }1081            }1082        }1083    }1084 1085    // pass 5: split graph, find tensors that need to be copied1086    {1087        int i_split = 0;1088        struct ggml_backend_sched_split * split = &sched->splits[0];1089        // find the backend of the first split, skipping view ops1090        int i = 0;1091        for (; i < graph->n_nodes; i++) {1092            struct ggml_tensor * node = graph->nodes[i];1093            if (!ggml_is_view_op(node->op)) {1094                split->backend_id = tensor_backend_id(node);1095                break;1096            }1097        }1098        split->i_start = 0;1099        split->n_inputs = 0;1100        int cur_backend_id = split->backend_id;1101        for (; i < graph->n_nodes; i++) {1102            struct ggml_tensor * node = graph->nodes[i];1103 1104            if (ggml_is_view_op(node->op)) {1105                continue;1106            }1107 1108            const int node_backend_id = tensor_backend_id(node);1109 1110            assert(node_backend_id != -1); // all nodes should be assigned by now1111 1112            // check if we should start a new split based on the sources of the current node1113            bool need_new_split = false;1114            if (node_backend_id == cur_backend_id && split->n_inputs > 0) {1115                for (int j = 0; j < GGML_MAX_SRC; j++) {1116                    struct ggml_tensor * src = node->src[j];1117                    if (src == NULL) {1118                        continue;1119                    }1120                    // check if a weight is on a different and incompatible backend1121                    // by starting a new split, the memory of the previously offloaded weights can be reused1122                    if (src->buffer != NULL && src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {1123                        int src_backend_id = tensor_backend_id(src);1124                        if (src_backend_id != cur_backend_id && !ggml_backend_sched_buffer_supported(sched, src, cur_backend_id)) {1125                            need_new_split = true;1126                            break;1127                        }1128                    }1129                    // check if the split has too many inputs1130                    // FIXME: count the number of inputs instead of only checking when full1131                    if (split->n_inputs == GGML_SCHED_MAX_SPLIT_INPUTS) {1132                        const size_t id = hash_id(src);1133                        int src_backend_id = sched->hv_tensor_backend_ids[id];1134                        bool supported = ggml_backend_sched_buffer_supported(sched, src, cur_backend_id);1135                        if (src_backend_id != cur_backend_id && tensor_id_copy(id, cur_backend_id, 0) == NULL && !supported) {1136                            need_new_split = true;1137                            break;1138                        }1139                    }1140                }1141            }1142 1143            if (node_backend_id != cur_backend_id || need_new_split) {1144                split->i_end = i;1145                i_split++;1146                if (i_split >= sched->splits_capacity) {1147                    sched->splits_capacity *= 2;1148                    sched->splits = (ggml_backend_sched_split *)1149                        realloc(sched->splits, sched->splits_capacity * sizeof(struct ggml_backend_sched_split));1150                    GGML_ASSERT(sched->splits != NULL);1151                }1152                split = &sched->splits[i_split];1153                split->backend_id = node_backend_id;1154                split->i_start = i;1155                split->n_inputs = 0;1156                cur_backend_id = node_backend_id;1157            }1158 1159            // find inputs that are not on the same backend1160            for (int j = 0; j < GGML_MAX_SRC; j++) {1161                struct ggml_tensor * src = node->src[j];1162                if (src == NULL) {1163                    continue;1164                }1165 1166                size_t src_id = hash_id(src);1167                const int src_backend_id = sched->hv_tensor_backend_ids[src_id];1168                assert(src_backend_id != -1); // all inputs should be assigned by now1169 1170                if (src->flags & GGML_TENSOR_FLAG_INPUT && sched->n_copies > 1) {1171                    if (tensor_id_copy(src_id, src_backend_id, 0) == NULL) {1172                        ggml_backend_t backend = sched->backends[src_backend_id];1173                        for (int c = 0; c < sched->n_copies; c++) {1174                            struct ggml_tensor * tensor_copy;1175                            if (c == sched->cur_copy) {1176                                tensor_copy = src; // use the original tensor as the current copy1177                            } else {1178                                tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);1179                                ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);1180                            }1181                            if (sched->n_copies > 1) {1182                                ggml_set_input(tensor_copy);1183                                ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor1184                            }1185                            tensor_id_copy(src_id, src_backend_id, c) = tensor_copy;1186                            SET_CAUSE(tensor_copy, "4.cpy");1187                        }1188                        int n_graph_inputs = sched->n_graph_inputs++;1189                        GGML_ASSERT(n_graph_inputs < GGML_SCHED_MAX_SPLIT_INPUTS);1190                        sched->graph_inputs[n_graph_inputs] = src;1191                    }1192                }1193 1194                if (src_backend_id != cur_backend_id && !ggml_backend_sched_buffer_supported(sched, src, cur_backend_id)) {1195                    // create a copy of the input in the split's backend1196                    if (tensor_id_copy(src_id, cur_backend_id, 0) == NULL) {1197                        ggml_backend_t backend = sched->backends[cur_backend_id];1198                        for (int c = 0; c < sched->n_copies; c++) {1199                            struct ggml_tensor * tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);1200                            ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);

Showing the first 1,200 of 2003 lines. Download the file for the rest.