KBaba7/llama.cpp
0
1// Note: porting this file to C++ is a work in progress2 3#ifdef _WIN324#define WIN32_LEAN_AND_MEAN5#ifndef NOMINMAX6# define NOMINMAX7#endif8#include <windows.h>9#endif10 11#include "ggml-backend.h"12#include "ggml-backend-impl.h"13#include "ggml-alloc.h"14#include "ggml-impl.h"15 16#include <assert.h>17#include <limits.h>18#include <stdarg.h>19#include <stdio.h>20#include <stdlib.h>21#include <string.h>22#include <string>23#include <vector>24 25#ifdef __APPLE__26#include <sys/types.h>27#include <sys/sysctl.h>28#endif29 30 31// backend buffer type32 33const char * ggml_backend_buft_name(ggml_backend_buffer_type_t buft) {34 return buft->iface.get_name(buft);35}36 37ggml_backend_buffer_t ggml_backend_buft_alloc_buffer(ggml_backend_buffer_type_t buft, size_t size) {38 if (size == 0) {39 // return a dummy buffer for zero-sized allocations40 return ggml_backend_buffer_init(buft, {}, NULL, 0);41 }42 43 return buft->iface.alloc_buffer(buft, size);44}45 46size_t ggml_backend_buft_get_alignment(ggml_backend_buffer_type_t buft) {47 return buft->iface.get_alignment(buft);48}49 50size_t ggml_backend_buft_get_max_size(ggml_backend_buffer_type_t buft) {51 // get_max_size is optional, defaults to SIZE_MAX52 if (buft->iface.get_max_size) {53 return buft->iface.get_max_size(buft);54 }55 return SIZE_MAX;56}57 58size_t ggml_backend_buft_get_alloc_size(ggml_backend_buffer_type_t buft, struct ggml_tensor * tensor) {59 // get_alloc_size is optional, defaults to ggml_nbytes60 if (buft->iface.get_alloc_size) {61 size_t size = buft->iface.get_alloc_size(buft, tensor);62 assert(size >= ggml_nbytes(tensor));63 return size;64 }65 return ggml_nbytes(tensor);66}67 68bool ggml_backend_buft_is_host(ggml_backend_buffer_type_t buft) {69 if (buft->iface.is_host) {70 return buft->iface.is_host(buft);71 }72 return false;73}74 75ggml_backend_dev_t ggml_backend_buft_get_device(ggml_backend_buffer_type_t buft) {76 return buft->device;77}78 79// backend buffer80 81ggml_backend_buffer_t ggml_backend_buffer_init(82 ggml_backend_buffer_type_t buft,83 struct ggml_backend_buffer_i iface,84 void * context,85 size_t size) {86 ggml_backend_buffer_t buffer = new ggml_backend_buffer {87 /* .interface = */ iface,88 /* .buft = */ buft,89 /* .context = */ context,90 /* .size = */ size,91 /* .usage = */ GGML_BACKEND_BUFFER_USAGE_ANY92 };93 94 return buffer;95}96 97const char * ggml_backend_buffer_name(ggml_backend_buffer_t buffer) {98 return ggml_backend_buft_name(ggml_backend_buffer_get_type(buffer));99}100 101void ggml_backend_buffer_free(ggml_backend_buffer_t buffer) {102 if (buffer == NULL) {103 return;104 }105 106 if (buffer->iface.free_buffer != NULL) {107 buffer->iface.free_buffer(buffer);108 }109 delete buffer;110}111 112size_t ggml_backend_buffer_get_size(ggml_backend_buffer_t buffer) {113 return buffer->size;114}115 116void * ggml_backend_buffer_get_base(ggml_backend_buffer_t buffer) {117 // get_base is optional if the buffer is zero-sized118 if (buffer->size == 0) {119 return NULL;120 }121 122 void * base = buffer->iface.get_base(buffer);123 124 GGML_ASSERT(base != NULL && "backend buffer base cannot be NULL");125 126 return base;127}128 129void ggml_backend_buffer_init_tensor(ggml_backend_buffer_t buffer, struct ggml_tensor * tensor) {130 // init_tensor is optional131 if (buffer->iface.init_tensor) {132 buffer->iface.init_tensor(buffer, tensor);133 }134}135 136void ggml_backend_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) {137 // clear is optional if the buffer is zero-sized138 if (buffer->size == 0) {139 return;140 }141 142 buffer->iface.clear(buffer, value);143}144 145size_t ggml_backend_buffer_get_alignment(ggml_backend_buffer_t buffer) {146 return ggml_backend_buft_get_alignment(ggml_backend_buffer_get_type(buffer));147}148 149size_t ggml_backend_buffer_get_max_size(ggml_backend_buffer_t buffer) {150 return ggml_backend_buft_get_max_size(ggml_backend_buffer_get_type(buffer));151}152 153size_t ggml_backend_buffer_get_alloc_size(ggml_backend_buffer_t buffer, struct ggml_tensor * tensor) {154 return ggml_backend_buft_get_alloc_size(ggml_backend_buffer_get_type(buffer), tensor);155}156 157bool ggml_backend_buffer_is_host(ggml_backend_buffer_t buffer) {158 return ggml_backend_buft_is_host(ggml_backend_buffer_get_type(buffer));159}160 161void ggml_backend_buffer_set_usage(ggml_backend_buffer_t buffer, enum ggml_backend_buffer_usage usage) {162 buffer->usage = usage;163 164 // FIXME: add a generic callback to the buffer interface165 if (ggml_backend_buffer_is_multi_buffer(buffer)) {166 ggml_backend_multi_buffer_set_usage(buffer, usage);167 }168}169 170enum ggml_backend_buffer_usage ggml_backend_buffer_get_usage(ggml_backend_buffer_t buffer) {171 return buffer->usage;172}173 174ggml_backend_buffer_type_t ggml_backend_buffer_get_type(ggml_backend_buffer_t buffer) {175 return buffer->buft;176}177 178void ggml_backend_buffer_reset(ggml_backend_buffer_t buffer) {179 if (buffer->iface.reset) {180 buffer->iface.reset(buffer);181 }182}183 184bool ggml_backend_buffer_copy_tensor(const struct ggml_tensor * src, struct ggml_tensor * dst) {185 ggml_backend_buffer_t dst_buf = dst->view_src ? dst->view_src->buffer : dst->buffer;186 if (dst_buf->iface.cpy_tensor) {187 return dst_buf->iface.cpy_tensor(dst_buf, src, dst);188 }189 return false;190}191 192// backend193 194ggml_guid_t ggml_backend_guid(ggml_backend_t backend) {195 if (backend == NULL) {196 return NULL;197 }198 return backend->guid;199}200 201const char * ggml_backend_name(ggml_backend_t backend) {202 if (backend == NULL) {203 return "NULL";204 }205 return backend->iface.get_name(backend);206}207 208void ggml_backend_free(ggml_backend_t backend) {209 if (backend == NULL) {210 return;211 }212 213 backend->iface.free(backend);214}215 216ggml_backend_buffer_type_t ggml_backend_get_default_buffer_type(ggml_backend_t backend) {217 return ggml_backend_dev_buffer_type(backend->device);218}219 220ggml_backend_buffer_t ggml_backend_alloc_buffer(ggml_backend_t backend, size_t size) {221 return ggml_backend_buft_alloc_buffer(ggml_backend_get_default_buffer_type(backend), size);222}223 224size_t ggml_backend_get_alignment(ggml_backend_t backend) {225 return ggml_backend_buft_get_alignment(ggml_backend_get_default_buffer_type(backend));226}227 228size_t ggml_backend_get_max_size(ggml_backend_t backend) {229 return ggml_backend_buft_get_max_size(ggml_backend_get_default_buffer_type(backend));230}231 232void ggml_backend_tensor_set_async(ggml_backend_t backend, struct ggml_tensor * tensor, const void * data, size_t offset, size_t size) {233 GGML_ASSERT(tensor->data != NULL && "tensor not allocated");234 GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor write out of bounds");235 236 if (backend->iface.set_tensor_async == NULL) {237 ggml_backend_tensor_set(tensor, data, offset, size);238 } else {239 backend->iface.set_tensor_async(backend, tensor, data, offset, size);240 }241}242 243void ggml_backend_tensor_get_async(ggml_backend_t backend, const struct ggml_tensor * tensor, void * data, size_t offset, size_t size) {244 GGML_ASSERT(tensor->data != NULL && "tensor not allocated");245 GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor read out of bounds");246 247 if (backend->iface.get_tensor_async == NULL) {248 ggml_backend_tensor_get(tensor, data, offset, size);249 } else {250 backend->iface.get_tensor_async(backend, tensor, data, offset, size);251 }252}253 254void ggml_backend_tensor_set(struct ggml_tensor * tensor, const void * data, size_t offset, size_t size) {255 GGML_ASSERT(tensor);256 ggml_backend_buffer_t buf = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;257 258 if (size == 0) {259 return;260 }261 262 GGML_ASSERT(buf != NULL && "tensor buffer not set");263 GGML_ASSERT(tensor->data != NULL && "tensor not allocated");264 GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor write out of bounds");265 266 buf->iface.set_tensor(buf, tensor, data, offset, size);267}268 269void ggml_backend_tensor_get(const struct ggml_tensor * tensor, void * data, size_t offset, size_t size) {270 GGML_ASSERT(tensor);271 ggml_backend_buffer_t buf = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;272 273 if (size == 0) {274 return;275 }276 277 GGML_ASSERT(buf != NULL && "tensor buffer not set");278 GGML_ASSERT(tensor->data != NULL && "tensor not allocated");279 GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor read out of bounds");280 281 buf->iface.get_tensor(buf, tensor, data, offset, size);282}283 284void ggml_backend_tensor_memset(struct ggml_tensor * tensor, uint8_t value, size_t offset, size_t size) {285 ggml_backend_buffer_t buf = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;286 287 if (size == 0) {288 return;289 }290 291 GGML_ASSERT(buf != NULL && "tensor buffer not set");292 GGML_ASSERT(tensor->data != NULL && "tensor not allocated");293 GGML_ASSERT(offset + size <= ggml_nbytes(tensor) && "tensor write out of bounds");294 GGML_ASSERT(buf->iface.memset_tensor != NULL && "memset not implemented by backend buffer");295 296 buf->iface.memset_tensor(buf, tensor, value, offset, size);297}298 299void ggml_backend_synchronize(ggml_backend_t backend) {300 if (backend->iface.synchronize == NULL) {301 return;302 }303 304 backend->iface.synchronize(backend);305}306 307ggml_backend_graph_plan_t ggml_backend_graph_plan_create(ggml_backend_t backend, struct ggml_cgraph * cgraph) {308 GGML_ASSERT(backend->iface.graph_plan_create != NULL);309 310 return backend->iface.graph_plan_create(backend, cgraph);311}312 313void ggml_backend_graph_plan_free(ggml_backend_t backend, ggml_backend_graph_plan_t plan) {314 GGML_ASSERT(backend->iface.graph_plan_free != NULL);315 316 backend->iface.graph_plan_free(backend, plan);317}318 319enum ggml_status ggml_backend_graph_plan_compute(ggml_backend_t backend, ggml_backend_graph_plan_t plan) {320 GGML_ASSERT(backend->iface.graph_plan_compute != NULL);321 322 return backend->iface.graph_plan_compute(backend, plan);323}324 325enum ggml_status ggml_backend_graph_compute(ggml_backend_t backend, struct ggml_cgraph * cgraph) {326 enum ggml_status err = ggml_backend_graph_compute_async(backend, cgraph);327 ggml_backend_synchronize(backend);328 return err;329}330 331enum ggml_status ggml_backend_graph_compute_async(ggml_backend_t backend, struct ggml_cgraph * cgraph) {332 return backend->iface.graph_compute(backend, cgraph);333}334 335bool ggml_backend_supports_op(ggml_backend_t backend, const struct ggml_tensor * op) {336 return ggml_backend_dev_supports_op(backend->device, op);337}338 339bool ggml_backend_supports_buft(ggml_backend_t backend, ggml_backend_buffer_type_t buft) {340 return ggml_backend_dev_supports_buft(backend->device, buft);341}342 343bool ggml_backend_offload_op(ggml_backend_t backend, const struct ggml_tensor * op) {344 return ggml_backend_dev_offload_op(backend->device, op);345}346 347ggml_backend_dev_t ggml_backend_get_device(ggml_backend_t backend) {348 return backend->device;349}350 351// backend copy352 353static bool ggml_are_same_layout(const struct ggml_tensor * a, const struct ggml_tensor * b) {354 if (a->type != b->type) {355 return false;356 }357 for (int i = 0; i < GGML_MAX_DIMS; i++) {358 if (a->ne[i] != b->ne[i]) {359 return false;360 }361 if (a->nb[i] != b->nb[i]) {362 return false;363 }364 }365 return true;366}367 368void ggml_backend_tensor_copy(struct ggml_tensor * src, struct ggml_tensor * dst) {369 GGML_ASSERT(ggml_are_same_layout(src, dst) && "cannot copy tensors with different layouts");370 371 if (src == dst) {372 return;373 }374 375 if (ggml_backend_buffer_is_host(src->buffer)) {376 ggml_backend_tensor_set(dst, src->data, 0, ggml_nbytes(src));377 } else if (ggml_backend_buffer_is_host(dst->buffer)) {378 ggml_backend_tensor_get(src, dst->data, 0, ggml_nbytes(src));379 } else if (!ggml_backend_buffer_copy_tensor(src, dst)) {380#ifndef NDEBUG381 GGML_LOG_DEBUG("%s: warning: slow copy from %s to %s\n", __func__, ggml_backend_buffer_name(src->buffer), ggml_backend_buffer_name(dst->buffer));382#endif383 size_t nbytes = ggml_nbytes(src);384 void * data = malloc(nbytes);385 ggml_backend_tensor_get(src, data, 0, nbytes);386 ggml_backend_tensor_set(dst, data, 0, nbytes);387 free(data);388 }389}390 391void ggml_backend_tensor_copy_async(ggml_backend_t backend_src, ggml_backend_t backend_dst, struct ggml_tensor * src, struct ggml_tensor * dst) {392 GGML_ASSERT(ggml_are_same_layout(src, dst) && "cannot copy tensors with different layouts");393 394 if (src == dst) {395 return;396 }397 398 if (backend_dst->iface.cpy_tensor_async != NULL) {399 if (backend_dst->iface.cpy_tensor_async(backend_src, backend_dst, src, dst)) {400 return;401 }402 }403 404 // an async copy would normally happen after all the queued operations on both backends are completed405 // to simulate the same behavior, we need to synchronize both backends first, and do a blocking copy406 ggml_backend_synchronize(backend_src);407 ggml_backend_synchronize(backend_dst);408 ggml_backend_tensor_copy(src, dst);409}410 411// events412 413ggml_backend_event_t ggml_backend_event_new(ggml_backend_dev_t device) {414 // null device is allowed for the transition period to the device interface415 if (device == NULL || device->iface.event_new == NULL) {416 return NULL;417 }418 return device->iface.event_new(device);419}420 421void ggml_backend_event_free(ggml_backend_event_t event) {422 if (event == NULL) {423 return;424 }425 event->device->iface.event_free(event->device, event);426}427 428void ggml_backend_event_record(ggml_backend_event_t event, ggml_backend_t backend) {429 GGML_ASSERT(backend->iface.event_record != NULL);430 431 backend->iface.event_record(backend, event);432}433 434void ggml_backend_event_synchronize(ggml_backend_event_t event) {435 GGML_ASSERT(event->device->iface.event_synchronize);436 437 event->device->iface.event_synchronize(event->device, event);438}439 440void ggml_backend_event_wait(ggml_backend_t backend, ggml_backend_event_t event) {441 GGML_ASSERT(backend->iface.event_wait != NULL);442 443 backend->iface.event_wait(backend, event);444}445 446// Backend device447 448const char * ggml_backend_dev_name(ggml_backend_dev_t device) {449 return device->iface.get_name(device);450}451 452const char * ggml_backend_dev_description(ggml_backend_dev_t device) {453 return device->iface.get_description(device);454}455 456void ggml_backend_dev_memory(ggml_backend_dev_t device, size_t * free, size_t * total) {457 device->iface.get_memory(device, free, total);458}459 460enum ggml_backend_dev_type ggml_backend_dev_type(ggml_backend_dev_t device) {461 return device->iface.get_type(device);462}463 464void ggml_backend_dev_get_props(ggml_backend_dev_t device, struct ggml_backend_dev_props * props) {465 memset(props, 0, sizeof(*props));466 device->iface.get_props(device, props);467}468 469ggml_backend_reg_t ggml_backend_dev_backend_reg(ggml_backend_dev_t device) {470 return device->reg;471}472 473ggml_backend_t ggml_backend_dev_init(ggml_backend_dev_t device, const char * params) {474 return device->iface.init_backend(device, params);475}476 477ggml_backend_buffer_type_t ggml_backend_dev_buffer_type(ggml_backend_dev_t device) {478 return device->iface.get_buffer_type(device);479}480 481ggml_backend_buffer_type_t ggml_backend_dev_host_buffer_type(ggml_backend_dev_t device) {482 if (device->iface.get_host_buffer_type == NULL) {483 return NULL;484 }485 486 return device->iface.get_host_buffer_type(device);487}488 489ggml_backend_buffer_t ggml_backend_dev_buffer_from_host_ptr(ggml_backend_dev_t device, void * ptr, size_t size, size_t max_tensor_size) {490 return device->iface.buffer_from_host_ptr(device, ptr, size, max_tensor_size);491}492 493bool ggml_backend_dev_supports_op(ggml_backend_dev_t device, const struct ggml_tensor * op) {494 return device->iface.supports_op(device, op);495}496 497bool ggml_backend_dev_supports_buft(ggml_backend_dev_t device, ggml_backend_buffer_type_t buft) {498 return device->iface.supports_buft(device, buft);499}500 501bool ggml_backend_dev_offload_op(ggml_backend_dev_t device, const struct ggml_tensor * op) {502 if (device->iface.offload_op != NULL) {503 return device->iface.offload_op(device, op);504 }505 506 return false;507}508 509// Backend (reg)510 511const char * ggml_backend_reg_name(ggml_backend_reg_t reg) {512 return reg->iface.get_name(reg);513}514 515size_t ggml_backend_reg_dev_count(ggml_backend_reg_t reg) {516 return reg->iface.get_device_count(reg);517}518 519ggml_backend_dev_t ggml_backend_reg_dev_get(ggml_backend_reg_t reg, size_t index) {520 return reg->iface.get_device(reg, index);521}522 523void * ggml_backend_reg_get_proc_address(ggml_backend_reg_t reg, const char * name) {524 if (!reg->iface.get_proc_address) {525 return NULL;526 }527 return reg->iface.get_proc_address(reg, name);528}529 530// multi-buffer buffer531 532struct ggml_backend_multi_buffer_context {533 ggml_backend_buffer_t * buffers;534 size_t n_buffers;535};536 537static void ggml_backend_multi_buffer_free_buffer(ggml_backend_buffer_t buffer) {538 ggml_backend_multi_buffer_context * ctx = (ggml_backend_multi_buffer_context *) buffer->context;539 for (size_t i = 0; i < ctx->n_buffers; i++) {540 ggml_backend_buffer_free(ctx->buffers[i]);541 }542 543 free(ctx->buffers);544 free(ctx);545}546 547static void ggml_backend_multi_buffer_clear(ggml_backend_buffer_t buffer, uint8_t value) {548 ggml_backend_multi_buffer_context * ctx = (ggml_backend_multi_buffer_context *) buffer->context;549 for (size_t i = 0; i < ctx->n_buffers; i++) {550 ggml_backend_buffer_clear(ctx->buffers[i], value);551 }552}553 554static const struct ggml_backend_buffer_i ggml_backend_multi_buffer_i = {555 /* .free_buffer = */ ggml_backend_multi_buffer_free_buffer,556 /* .get_base = */ NULL,557 /* .init_tensor = */ NULL,558 /* .memset_tensor = */ NULL,559 /* .set_tensor = */ NULL,560 /* .get_tensor = */ NULL,561 /* .cpy_tensor = */ NULL,562 /* .clear = */ ggml_backend_multi_buffer_clear,563 /* .reset = */ NULL,564};565 566ggml_backend_buffer_t ggml_backend_multi_buffer_alloc_buffer(ggml_backend_buffer_t * buffers, size_t n_buffers) {567 ggml_backend_multi_buffer_context * ctx = (ggml_backend_multi_buffer_context *) malloc(sizeof(struct ggml_backend_multi_buffer_context));568 ctx->n_buffers = n_buffers;569 ctx->buffers = (ggml_backend_buffer_t *) malloc(n_buffers * sizeof(ggml_backend_buffer_t));570 571 GGML_ASSERT(ctx->buffers != NULL);572 573 size_t total_size = 0;574 for (size_t i = 0; i < n_buffers; i++) {575 ctx->buffers[i] = buffers[i];576 total_size += ggml_backend_buffer_get_size(buffers[i]);577 }578 579 return ggml_backend_buffer_init(buffers[0]->buft, ggml_backend_multi_buffer_i, ctx, total_size);580}581 582bool ggml_backend_buffer_is_multi_buffer(ggml_backend_buffer_t buffer) {583 return buffer->iface.free_buffer == ggml_backend_multi_buffer_free_buffer;584}585 586void ggml_backend_multi_buffer_set_usage(ggml_backend_buffer_t buffer, enum ggml_backend_buffer_usage usage) {587 GGML_ASSERT(ggml_backend_buffer_is_multi_buffer(buffer));588 ggml_backend_multi_buffer_context * ctx = (ggml_backend_multi_buffer_context *) buffer->context;589 for (size_t i = 0; i < ctx->n_buffers; i++) {590 ggml_backend_buffer_set_usage(ctx->buffers[i], usage);591 }592}593 594// creates a copy of the tensor with the same memory layout595static struct ggml_tensor * ggml_dup_tensor_layout(struct ggml_context * ctx, const struct ggml_tensor * tensor) {596 struct ggml_tensor * dup = ggml_dup_tensor(ctx, tensor);597 for (int i = 0; i < GGML_MAX_DIMS; i++) {598 dup->nb[i] = tensor->nb[i];599 }600 return dup;601}602 603static bool ggml_is_view_op(enum ggml_op op) {604 return op == GGML_OP_VIEW || op == GGML_OP_RESHAPE || op == GGML_OP_PERMUTE || op == GGML_OP_TRANSPOSE;605}606 607// scheduler608 609#ifndef GGML_SCHED_MAX_BACKENDS610#define GGML_SCHED_MAX_BACKENDS 16611#endif612 613#ifndef GGML_SCHED_MAX_SPLIT_INPUTS614#define GGML_SCHED_MAX_SPLIT_INPUTS GGML_MAX_SRC615#endif616 617#ifndef GGML_SCHED_MAX_COPIES618#define GGML_SCHED_MAX_COPIES 4619#endif620 621struct ggml_backend_sched_split {622 int backend_id;623 int i_start;624 int i_end;625 struct ggml_tensor * inputs[GGML_SCHED_MAX_SPLIT_INPUTS];626 int n_inputs;627 // graph view of this split628 struct ggml_cgraph graph;629};630 631struct ggml_backend_sched {632 bool is_reset; // true if the scheduler has been reset since the last graph split633 bool is_alloc;634 635 int n_backends;636 637 ggml_backend_t backends[GGML_SCHED_MAX_BACKENDS];638 ggml_backend_buffer_type_t bufts[GGML_SCHED_MAX_BACKENDS];639 ggml_gallocr_t galloc;640 641 // hash map of the nodes in the graph642 struct ggml_hash_set hash_set;643 int * hv_tensor_backend_ids; // [hash_set.size]644 struct ggml_tensor ** hv_tensor_copies; // [hash_set.size][n_backends][n_copies]645 646 int * node_backend_ids; // [graph_size]647 int * leaf_backend_ids; // [graph_size]648 649 int * prev_node_backend_ids; // [graph_size]650 int * prev_leaf_backend_ids; // [graph_size]651 652 // copy of the graph with modified inputs653 struct ggml_cgraph graph;654 655 // graph splits656 struct ggml_backend_sched_split * splits;657 int n_splits;658 int splits_capacity;659 660 // pipeline parallelism support661 int n_copies;662 int cur_copy;663 ggml_backend_event_t events[GGML_SCHED_MAX_BACKENDS][GGML_SCHED_MAX_COPIES];664 struct ggml_tensor * graph_inputs[GGML_SCHED_MAX_SPLIT_INPUTS];665 int n_graph_inputs;666 667 struct ggml_context * ctx;668 669 ggml_backend_sched_eval_callback callback_eval;670 void * callback_eval_user_data;671 672 char * context_buffer;673 size_t context_buffer_size;674 675 int debug;676};677 678#define hash_id(tensor) ggml_hash_find_or_insert(&sched->hash_set, tensor)679#define tensor_backend_id(tensor) sched->hv_tensor_backend_ids[hash_id(tensor)]680#define tensor_id_copy(id, backend_id, copy_id) sched->hv_tensor_copies[(id) * sched->n_backends * sched->n_copies + (backend_id) * sched->n_copies + (copy_id)]681#define tensor_copy(tensor, backend_id, copy_id) tensor_id_copy(hash_id(tensor), backend_id, copy_id)682 683// returns the priority of the backend, lower id is higher priority684static int ggml_backend_sched_backend_id(ggml_backend_sched_t sched, ggml_backend_t backend) {685 for (int i = 0; i < sched->n_backends; i++) {686 if (sched->backends[i] == backend) {687 return i;688 }689 }690 return -1;691}692 693static int ggml_backend_sched_backend_from_buffer(ggml_backend_sched_t sched, const struct ggml_tensor * tensor, const struct ggml_tensor * op) {694 ggml_backend_buffer_t buffer = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;695 if (buffer == NULL) {696 return -1;697 }698 699 // find highest prio backend that supports the buffer type and the op700 for (int i = 0; i < sched->n_backends; i++) {701 if (ggml_backend_supports_buft(sched->backends[i], buffer->buft) &&702 ggml_backend_supports_op(sched->backends[i], op)) {703 return i;704 }705 }706 707#ifndef NDEBUG708 GGML_LOG_DEBUG("%s: warning: no backend supports op %s with a weight with buffer type %s used in tensor %s, the weight will need to be copied\n",709 __func__, ggml_op_desc(tensor), ggml_backend_buffer_name(buffer), tensor->name);710#endif711 712 return -1;713}714 715#if 0716#define GGML_SCHED_MAX_SPLITS_DEBUG 4096717static char causes[GGML_DEFAULT_GRAPH_SIZE*16 + GGML_SCHED_MAX_SPLITS_DEBUG*GGML_SCHED_MAX_SPLIT_INPUTS][128]; // debug only718#define SET_CAUSE(node, ...) sprintf(causes[hash_id(node)], __VA_ARGS__)719#define GET_CAUSE(node) causes[hash_id(node)]720#else721#define SET_CAUSE(node, ...)722#define GET_CAUSE(node) ""723#endif724 725// returns the backend that should be used for the node based on the current locations726static int ggml_backend_sched_backend_id_from_cur(ggml_backend_sched_t sched, struct ggml_tensor * tensor) {727 // assign pre-allocated nodes to their backend728 int cur_backend_id = ggml_backend_sched_backend_from_buffer(sched, tensor, tensor);729 if (cur_backend_id != -1) {730 SET_CAUSE(tensor, "1.dst");731 return cur_backend_id;732 }733 734 // view_src735 if (tensor->view_src != NULL) {736 cur_backend_id = ggml_backend_sched_backend_from_buffer(sched, tensor->view_src, tensor);737 if (cur_backend_id != -1) {738 SET_CAUSE(tensor, "1.vsrc");739 return cur_backend_id;740 }741 }742 743 if (tensor->buffer || (tensor->view_src && tensor->view_src->buffer)) {744 // since the tensor is pre-allocated, it cannot be moved to another backend745 ggml_backend_buffer_t buffer = tensor->view_src ? tensor->view_src->buffer : tensor->buffer;746 GGML_ABORT("pre-allocated tensor (%s) in a buffer (%s) that cannot run the operation (%s)", tensor->name, ggml_backend_buffer_name(buffer), ggml_op_name(tensor->op));747 }748 749 // graph input750 if (tensor->flags & GGML_TENSOR_FLAG_INPUT) {751 cur_backend_id = sched->n_backends - 1; // last backend (assumed CPU)752 SET_CAUSE(tensor, "1.inp");753 return cur_backend_id;754 }755 756 // operations with weights are preferably run on the same backend as the weights757 for (int i = 0; i < GGML_MAX_SRC; i++) {758 const struct ggml_tensor * src = tensor->src[i];759 if (src == NULL) {760 continue;761 }762 // skip ROPE since the rope freqs tensor is too small to choose a backend based on it763 // not an ideal solution764 if (tensor->op != GGML_OP_ROPE && src->buffer != NULL && src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {765 int src_backend_id = ggml_backend_sched_backend_from_buffer(sched, src, tensor);766 // check if a backend with higher prio wants to offload the op767 if (src_backend_id == sched->n_backends - 1 && ggml_backend_buffer_is_host(src->buffer)) {768 for (int b = 0; b < src_backend_id; b++) {769 if (ggml_backend_supports_op(sched->backends[b], tensor) && ggml_backend_offload_op(sched->backends[b], tensor)) {770 SET_CAUSE(tensor, "1.off");771 return b;772 }773 }774 }775 SET_CAUSE(tensor, "1.wgt%d", i);776 return src_backend_id;777 }778 }779 780 return -1;781}782 783static char * fmt_size(size_t size) {784 static char buffer[128];785 if (size >= 1024*1024) {786 snprintf(buffer, sizeof(buffer), "%zuM", size/1024/1024);787 } else {788 snprintf(buffer, sizeof(buffer), "%zuK", size/1024);789 }790 return buffer;791}792 793static void ggml_backend_sched_print_assignments(ggml_backend_sched_t sched, struct ggml_cgraph * graph) {794 int cur_split = 0;795 for (int i = 0; i < graph->n_nodes; i++) {796 if (cur_split < sched->n_splits && i == sched->splits[cur_split].i_start) {797 ggml_backend_t split_backend = sched->backends[sched->splits[cur_split].backend_id];798 GGML_LOG_DEBUG("\n## SPLIT #%d: %s # %d inputs", cur_split, ggml_backend_name(split_backend),799 sched->splits[cur_split].n_inputs);800 for (int j = 0; j < sched->splits[cur_split].n_inputs; j++) {801 if (j == 0) {802 GGML_LOG_DEBUG(": ");803 }804 GGML_LOG_DEBUG("[%s (%5.5s)] ", sched->splits[cur_split].inputs[j]->name,805 fmt_size(ggml_nbytes(sched->splits[cur_split].inputs[j])));806 }807 GGML_LOG_DEBUG("\n");808 cur_split++;809 }810 struct ggml_tensor * node = graph->nodes[i];811 if (ggml_is_view_op(node->op)) {812 continue;813 }814 if (sched->debug > 1) {815 ggml_backend_t tensor_backend = ggml_backend_sched_get_tensor_backend(sched, node);816 GGML_LOG_DEBUG("node #%3d (%10.10s): %20.20s (%5.5s) [%5.5s %8.8s]:", i, ggml_op_name(node->op), node->name,817 fmt_size(ggml_nbytes(node)), tensor_backend ? ggml_backend_name(tensor_backend) : "NULL", GET_CAUSE(node));818 for (int j = 0; j < GGML_MAX_SRC; j++) {819 struct ggml_tensor * src = node->src[j];820 if (src == NULL) {821 continue;822 }823 ggml_backend_t src_backend = ggml_backend_sched_get_tensor_backend(sched, src);824 GGML_LOG_DEBUG(" %20.20s (%5.5s) [%5.5s %8.8s]", src->name,825 fmt_size(ggml_nbytes(src)), src_backend ? ggml_backend_name(src_backend) : "NULL", GET_CAUSE(src));826 }827 GGML_LOG_DEBUG("\n");828 }829 }830}831 832static bool ggml_backend_sched_buffer_supported(ggml_backend_sched_t sched, struct ggml_tensor * t, int backend_id) {833 ggml_backend_buffer_t buf = t->view_src ? t->view_src->buffer : t->buffer;834 ggml_backend_buffer_type_t buft = NULL;835 836 if (buf) {837 // the tensor is already allocated838 buft = buf->buft;839 } else {840 // see if the tensor already has a backend assigned, and use the buffer type of that backend841 int tensor_backend_id = tensor_backend_id(t);842 if (tensor_backend_id == -1 && t->view_src) {843 tensor_backend_id = tensor_backend_id(t->view_src);844 }845 if (tensor_backend_id != -1) {846 buft = sched->bufts[tensor_backend_id];847 }848 }849 850 return buft != NULL && ggml_backend_supports_buft(sched->backends[backend_id], buft);851}852 853static void ggml_backend_sched_set_if_supported(ggml_backend_sched_t sched, struct ggml_tensor * node, int cur_backend_id, int * node_backend_id) {854 if (ggml_backend_supports_op(sched->backends[cur_backend_id], node)) {855 *node_backend_id = cur_backend_id;856 SET_CAUSE(node, "2.sup");857 }858}859 860// assigns backends to ops and splits the graph into subgraphs that can be computed on the same backend861static void ggml_backend_sched_split_graph(ggml_backend_sched_t sched, struct ggml_cgraph * graph) {862 // reset splits863 sched->n_splits = 0;864 sched->n_graph_inputs = 0;865 sched->is_reset = false;866 867 struct ggml_init_params params = {868 /* .mem_size = */ sched->context_buffer_size,869 /* .mem_buffer = */ sched->context_buffer,870 /* .no_alloc = */ true871 };872 873 ggml_free(sched->ctx);874 875 sched->ctx = ggml_init(params);876 if (sched->ctx == NULL) {877 GGML_ABORT("%s: failed to initialize context\n", __func__);878 }879 880 // pass 1: assign backends to ops with pre-allocated inputs881 for (int i = 0; i < graph->n_leafs; i++) {882 struct ggml_tensor * leaf = graph->leafs[i];883 int * leaf_backend_id = &tensor_backend_id(leaf);884 // do not overwrite user assignments885 if (*leaf_backend_id == -1) {886 *leaf_backend_id = ggml_backend_sched_backend_id_from_cur(sched, leaf);887 }888 }889 890 for (int i = 0; i < graph->n_nodes; i++) {891 struct ggml_tensor * node = graph->nodes[i];892 int * node_backend_id = &tensor_backend_id(node);893 // do not overwrite user assignments894 if (*node_backend_id == -1) {895 *node_backend_id = ggml_backend_sched_backend_id_from_cur(sched, node);896 897#if 0898 // src899 if (node->op == GGML_OP_NONE) {900 continue;901 }902 903 for (int j = 0; j < GGML_MAX_SRC; j++) {904 struct ggml_tensor * src = node->src[j];905 if (src == NULL) {906 continue;907 }908 int * src_backend_id = &tensor_backend_id(src);909 if (*src_backend_id == -1) {910 *src_backend_id = ggml_backend_sched_backend_id_from_cur(sched, src);911 }912 }913#endif914 }915 }916 917 // pass 2: expand current backend assignments918 // assign the same backend to adjacent nodes919 // expand gpu backends (i.e. non last prio) up and down, ignoring cpu (the lowest priority backend)920 // thus, cpu will never be used unless weights are on cpu, or there are no gpu ops between cpu ops921 // ops unsupported by the backend being expanded will be left unassigned so that they can be assigned later when the locations of its inputs are known922 // expand gpu down923 {924 int cur_backend_id = -1;925 for (int i = 0; i < graph->n_nodes; i++) {926 struct ggml_tensor * node = graph->nodes[i];927 if (ggml_is_view_op(node->op)) {928 continue;929 }930 int * node_backend_id = &tensor_backend_id(node);931 if (*node_backend_id != -1) {932 if (*node_backend_id == sched->n_backends - 1) {933 // skip cpu (lowest prio backend)934 cur_backend_id = -1;935 } else {936 cur_backend_id = *node_backend_id;937 }938 } else if (cur_backend_id != -1) {939 ggml_backend_sched_set_if_supported(sched, node, cur_backend_id, node_backend_id);940 }941 }942 }943 // expand gpu up944 {945 int cur_backend_id = -1;946 for (int i = graph->n_nodes - 1; i >= 0; i--) {947 struct ggml_tensor * node = graph->nodes[i];948 if (ggml_is_view_op(node->op)) {949 continue;950 }951 int * node_backend_id = &tensor_backend_id(node);952 if (*node_backend_id != -1) {953 if (*node_backend_id == sched->n_backends - 1) {954 // skip cpu (lowest prio backend)955 cur_backend_id = -1;956 } else {957 cur_backend_id = *node_backend_id;958 }959 } else if (cur_backend_id != -1) {960 ggml_backend_sched_set_if_supported(sched, node, cur_backend_id, node_backend_id);961 }962 }963 }964 // expand rest down965 {966 int cur_backend_id = -1;967 for (int i = 0; i < graph->n_nodes; i++) {968 struct ggml_tensor * node = graph->nodes[i];969 if (ggml_is_view_op(node->op)) {970 continue;971 }972 int * node_backend_id = &tensor_backend_id(node);973 if (*node_backend_id != -1) {974 cur_backend_id = *node_backend_id;975 } else if (cur_backend_id != -1) {976 ggml_backend_sched_set_if_supported(sched, node, cur_backend_id, node_backend_id);977 }978 }979 }980 // expand rest up981 {982 int cur_backend_id = -1;983 for (int i = graph->n_nodes - 1; i >= 0; i--) {984 struct ggml_tensor * node = graph->nodes[i];985 if (ggml_is_view_op(node->op)) {986 continue;987 }988 int * node_backend_id = &tensor_backend_id(node);989 if (*node_backend_id != -1) {990 cur_backend_id = *node_backend_id;991 } else if (cur_backend_id != -1) {992 ggml_backend_sched_set_if_supported(sched, node, cur_backend_id, node_backend_id);993 }994 }995 }996 997 // pass 3: upgrade nodes to higher prio backends with compatible buffer types998 // if the tensor is already in the same buffer type (*) as another higher priority backend, we should move it there999 // however, we also need to verify that the sources are in compatible buffer types1000 // (*) the actual requirement is more relaxed, the buffer type of the backend should be supported by all the users of this tensor further down the graph1001 // however, this is slow to verify, so we have a more strict requirement that the buffer type is the same1002 // this is not uncommon since multiple backends can use host memory, with the same buffer type (eg. BLAS and CPU)1003 // additionally, set remaining unassigned nodes to the backend with the most supported inputs1004 // only nodes that could not be assigned during expansion due to the backend not supporting the op should be unassigned at this point1005 for (int i = 0; i < graph->n_nodes; i++) {1006 struct ggml_tensor * node = graph->nodes[i];1007 if (ggml_is_view_op(node->op)) {1008 continue;1009 }1010 int * node_backend_id = &tensor_backend_id(node);1011 if (*node_backend_id == -1) {1012 // unassigned node: find the backend with the most supported inputs1013 int n_supported_best = -1;1014 for (int b = 0; b < sched->n_backends; b++) {1015 if (ggml_backend_supports_op(sched->backends[b], node)) {1016 int n_supported = 0;1017 for (int j = 0; j < GGML_MAX_SRC; j++) {1018 struct ggml_tensor * src = node->src[j];1019 if (src == NULL) {1020 continue;1021 }1022 if ((tensor_backend_id(src) != -1 || tensor_backend_id(src->view_src) != -1) && ggml_backend_sched_buffer_supported(sched, src, b)) {1023 n_supported++;1024 }1025 }1026 if (n_supported > n_supported_best) {1027 n_supported_best = n_supported;1028 *node_backend_id = b;1029 SET_CAUSE(node, "3.best");1030 }1031 }1032 }1033 } else {1034 // assigned node: upgrade to higher prio backend if possible1035 for (int b = 0; b < *node_backend_id; b++) {1036 if (sched->bufts[b] == sched->bufts[*node_backend_id] && ggml_backend_supports_op(sched->backends[b], node)) {1037 bool supported = true;1038 for (int j = 0; j < GGML_MAX_SRC; j++) {1039 struct ggml_tensor * src = node->src[j];1040 if (src == NULL) {1041 continue;1042 }1043 if (!ggml_backend_sched_buffer_supported(sched, src, b)) {1044 supported = false;1045 break;1046 }1047 }1048 if (supported) {1049 *node_backend_id = b;1050 SET_CAUSE(node, "3.upg");1051 break;1052 }1053 }1054 }1055 }1056 }1057 1058 // pass 4: assign backends to remaining src from dst and view_src1059 for (int i = 0; i < graph->n_nodes; i++) {1060 struct ggml_tensor * node = graph->nodes[i];1061 int * cur_backend_id = &tensor_backend_id(node);1062 if (node->view_src != NULL && *cur_backend_id == -1) {1063 *cur_backend_id = tensor_backend_id(node->view_src);1064 SET_CAUSE(node, "4.vsrc");1065 }1066 for (int j = 0; j < GGML_MAX_SRC; j++) {1067 struct ggml_tensor * src = node->src[j];1068 if (src == NULL) {1069 continue;1070 }1071 int * src_backend_id = &tensor_backend_id(src);1072 if (*src_backend_id == -1) {1073 if (src->view_src != NULL) {1074 // views are always on the same backend as the source1075 *src_backend_id = tensor_backend_id(src->view_src);1076 SET_CAUSE(src, "4.vsrc");1077 } else {1078 *src_backend_id = *cur_backend_id;1079 SET_CAUSE(src, "4.cur");1080 }1081 }1082 }1083 }1084 1085 // pass 5: split graph, find tensors that need to be copied1086 {1087 int i_split = 0;1088 struct ggml_backend_sched_split * split = &sched->splits[0];1089 // find the backend of the first split, skipping view ops1090 int i = 0;1091 for (; i < graph->n_nodes; i++) {1092 struct ggml_tensor * node = graph->nodes[i];1093 if (!ggml_is_view_op(node->op)) {1094 split->backend_id = tensor_backend_id(node);1095 break;1096 }1097 }1098 split->i_start = 0;1099 split->n_inputs = 0;1100 int cur_backend_id = split->backend_id;1101 for (; i < graph->n_nodes; i++) {1102 struct ggml_tensor * node = graph->nodes[i];1103 1104 if (ggml_is_view_op(node->op)) {1105 continue;1106 }1107 1108 const int node_backend_id = tensor_backend_id(node);1109 1110 assert(node_backend_id != -1); // all nodes should be assigned by now1111 1112 // check if we should start a new split based on the sources of the current node1113 bool need_new_split = false;1114 if (node_backend_id == cur_backend_id && split->n_inputs > 0) {1115 for (int j = 0; j < GGML_MAX_SRC; j++) {1116 struct ggml_tensor * src = node->src[j];1117 if (src == NULL) {1118 continue;1119 }1120 // check if a weight is on a different and incompatible backend1121 // by starting a new split, the memory of the previously offloaded weights can be reused1122 if (src->buffer != NULL && src->buffer->usage == GGML_BACKEND_BUFFER_USAGE_WEIGHTS) {1123 int src_backend_id = tensor_backend_id(src);1124 if (src_backend_id != cur_backend_id && !ggml_backend_sched_buffer_supported(sched, src, cur_backend_id)) {1125 need_new_split = true;1126 break;1127 }1128 }1129 // check if the split has too many inputs1130 // FIXME: count the number of inputs instead of only checking when full1131 if (split->n_inputs == GGML_SCHED_MAX_SPLIT_INPUTS) {1132 const size_t id = hash_id(src);1133 int src_backend_id = sched->hv_tensor_backend_ids[id];1134 bool supported = ggml_backend_sched_buffer_supported(sched, src, cur_backend_id);1135 if (src_backend_id != cur_backend_id && tensor_id_copy(id, cur_backend_id, 0) == NULL && !supported) {1136 need_new_split = true;1137 break;1138 }1139 }1140 }1141 }1142 1143 if (node_backend_id != cur_backend_id || need_new_split) {1144 split->i_end = i;1145 i_split++;1146 if (i_split >= sched->splits_capacity) {1147 sched->splits_capacity *= 2;1148 sched->splits = (ggml_backend_sched_split *)1149 realloc(sched->splits, sched->splits_capacity * sizeof(struct ggml_backend_sched_split));1150 GGML_ASSERT(sched->splits != NULL);1151 }1152 split = &sched->splits[i_split];1153 split->backend_id = node_backend_id;1154 split->i_start = i;1155 split->n_inputs = 0;1156 cur_backend_id = node_backend_id;1157 }1158 1159 // find inputs that are not on the same backend1160 for (int j = 0; j < GGML_MAX_SRC; j++) {1161 struct ggml_tensor * src = node->src[j];1162 if (src == NULL) {1163 continue;1164 }1165 1166 size_t src_id = hash_id(src);1167 const int src_backend_id = sched->hv_tensor_backend_ids[src_id];1168 assert(src_backend_id != -1); // all inputs should be assigned by now1169 1170 if (src->flags & GGML_TENSOR_FLAG_INPUT && sched->n_copies > 1) {1171 if (tensor_id_copy(src_id, src_backend_id, 0) == NULL) {1172 ggml_backend_t backend = sched->backends[src_backend_id];1173 for (int c = 0; c < sched->n_copies; c++) {1174 struct ggml_tensor * tensor_copy;1175 if (c == sched->cur_copy) {1176 tensor_copy = src; // use the original tensor as the current copy1177 } else {1178 tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);1179 ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);1180 }1181 if (sched->n_copies > 1) {1182 ggml_set_input(tensor_copy);1183 ggml_set_output(tensor_copy); // prevent ggml-alloc from overwriting the tensor1184 }1185 tensor_id_copy(src_id, src_backend_id, c) = tensor_copy;1186 SET_CAUSE(tensor_copy, "4.cpy");1187 }1188 int n_graph_inputs = sched->n_graph_inputs++;1189 GGML_ASSERT(n_graph_inputs < GGML_SCHED_MAX_SPLIT_INPUTS);1190 sched->graph_inputs[n_graph_inputs] = src;1191 }1192 }1193 1194 if (src_backend_id != cur_backend_id && !ggml_backend_sched_buffer_supported(sched, src, cur_backend_id)) {1195 // create a copy of the input in the split's backend1196 if (tensor_id_copy(src_id, cur_backend_id, 0) == NULL) {1197 ggml_backend_t backend = sched->backends[cur_backend_id];1198 for (int c = 0; c < sched->n_copies; c++) {1199 struct ggml_tensor * tensor_copy = ggml_dup_tensor_layout(sched->ctx, src);1200 ggml_format_name(tensor_copy, "%s#%s#%d", ggml_backend_name(backend), src->name, c);