Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
ggml-alloc.c1249 linesDownload Raw Back to src
1#include "ggml-alloc.h"2#include "ggml-backend-impl.h"3#include "ggml.h"4#include "ggml-impl.h"5 6#include <assert.h>7#include <limits.h>8#include <stdarg.h>9#include <stdio.h>10#include <stdlib.h>11#include <string.h>12 13#define MAX(a, b) ((a) > (b) ? (a) : (b))14#define MAX_FREE_BLOCKS 25615 16//#define GGML_ALLOCATOR_DEBUG17 18//#define AT_PRINTF(...) GGML_LOG_DEBUG(__VA_ARGS__)19#define AT_PRINTF(...)20 21// ops that return true for this function must not use restrict pointers for their backend implementations22bool ggml_op_can_inplace(enum ggml_op op) {23    switch (op) {24        case GGML_OP_FILL:25        case GGML_OP_SCALE:26        case GGML_OP_DIAG_MASK_ZERO:27        case GGML_OP_DIAG_MASK_INF:28        case GGML_OP_ADD:29        case GGML_OP_ADD_ID:30        case GGML_OP_ADD1:31        case GGML_OP_SUB:32        case GGML_OP_MUL:33        case GGML_OP_DIV:34        case GGML_OP_SQR:35        case GGML_OP_SQRT:36        case GGML_OP_LOG:37        case GGML_OP_UNARY:38        case GGML_OP_ROPE:39        case GGML_OP_ROPE_BACK:40        case GGML_OP_SILU_BACK:41        case GGML_OP_RMS_NORM:42        case GGML_OP_RMS_NORM_BACK:43        case GGML_OP_SOFT_MAX:44        case GGML_OP_SOFT_MAX_BACK:45            return true;46 47        default:48            return false;49    }50}51 52static size_t aligned_offset(const void * buffer, size_t offset, size_t alignment) {53    assert(alignment && !(alignment & (alignment - 1))); // power of 254    size_t align = (alignment - (((uintptr_t)buffer + offset) % alignment)) % alignment;55    return offset + align;56}57 58// tallocr59 60struct ggml_tallocr ggml_tallocr_new(ggml_backend_buffer_t buffer) {61    void * base = ggml_backend_buffer_get_base(buffer);62    size_t align = ggml_backend_buffer_get_alignment(buffer);63 64    assert(align && !(align & (align - 1))); // power of 265 66    struct ggml_tallocr talloc = (struct ggml_tallocr) {67        /*.buffer    = */ buffer,68        /*.base      = */ base,69        /*.alignment = */ align,70        /*.offset    = */ aligned_offset(base, 0, align),71    };72    return talloc;73}74 75enum ggml_status ggml_tallocr_alloc(struct ggml_tallocr * talloc, struct ggml_tensor * tensor) {76    size_t size = ggml_backend_buffer_get_alloc_size(talloc->buffer, tensor);77    size = GGML_PAD(size, talloc->alignment);78 79    if (talloc->offset + size > ggml_backend_buffer_get_size(talloc->buffer)) {80        GGML_LOG_ERROR("%s: not enough space in the buffer to allocate %s (needed %zu, available %zu)\n",81                __func__, tensor->name, size, ggml_backend_buffer_get_size(talloc->buffer) - talloc->offset);82        GGML_ABORT("not enough space in the buffer");83    }84 85    void * addr = (char *)ggml_backend_buffer_get_base(talloc->buffer) + talloc->offset;86    talloc->offset += size;87 88    assert(((uintptr_t)addr % talloc->alignment) == 0);89 90    return ggml_backend_tensor_alloc(talloc->buffer, tensor, addr);91}92 93// dynamic tensor allocator94 95#define GGML_VBUFFER_MAX_CHUNKS 1696 97// relative memory address within an allocation that can be split into multiple buffers (chunks)98struct buffer_address {99    int chunk;     // index of a backend buffer100    size_t offset; // local memory offset within the buffer101};102 103static const struct buffer_address GGML_BUFFER_ADDRESS_INVALID = { -1, SIZE_MAX };104 105static bool ggml_buffer_address_less(struct buffer_address a, struct buffer_address b) {106    return a.chunk != b.chunk ? a.chunk < b.chunk : a.offset < b.offset;107}108 109struct free_block {110    size_t offset;111    size_t size;112};113 114struct tallocr_chunk {115    struct free_block free_blocks[MAX_FREE_BLOCKS];116    int n_free_blocks;117    size_t max_size;118};119 120struct ggml_dyn_tallocr {121    size_t alignment;122    size_t max_chunk_size;123    struct tallocr_chunk * chunks[GGML_VBUFFER_MAX_CHUNKS];124    int n_chunks;125 126#ifdef GGML_ALLOCATOR_DEBUG127    struct {128        const struct ggml_tensor * tensor;129        struct buffer_address addr;130    } allocated_tensors[1024];131#endif132};133 134static void ggml_dyn_tallocr_insert_block(struct tallocr_chunk * chunk, size_t offset, size_t size) {135    GGML_ASSERT(chunk->n_free_blocks < MAX_FREE_BLOCKS && "out of free blocks");136    // insert the new block in the correct position to keep the array sorted by address (to make merging blocks faster)137    int insert_pos = 0;138    while (insert_pos < chunk->n_free_blocks && chunk->free_blocks[insert_pos].offset < offset) {139        insert_pos++;140    }141    // shift all blocks from insert_pos onward to make room for the new block142    for (int i = chunk->n_free_blocks; i > insert_pos; i--) {143        chunk->free_blocks[i] = chunk->free_blocks[i-1];144    }145    // insert the new block146    chunk->free_blocks[insert_pos].offset = offset;147    chunk->free_blocks[insert_pos].size = size;148    chunk->n_free_blocks++;149}150 151static void ggml_dyn_tallocr_remove_block(struct tallocr_chunk * chunk, int idx) {152    // shift all elements after idx by 1 to the left, overwriting the element at idx153    for (int i = idx; i < chunk->n_free_blocks - 1; i++) {154        chunk->free_blocks[i] = chunk->free_blocks[i+1];155    }156    chunk->n_free_blocks--;157}158 159static int ggml_dyn_tallocr_new_chunk(struct ggml_dyn_tallocr * alloc, size_t min_size) {160    if (alloc->n_chunks >= GGML_VBUFFER_MAX_CHUNKS) {161        return -1;162    }163    struct tallocr_chunk * chunk = calloc(1, sizeof(struct tallocr_chunk));164    chunk->n_free_blocks = 1;165    chunk->free_blocks[0].offset = 0;166    // available space in a chunk is limited to max_chunk_size, but can be higher if:167    // 1. a single tensor exceeds the maximum, and cannot fit any other way168    // 2. we are running out of chunks169    // backends will either manage to allocate the larger size, or report an error.170    chunk->free_blocks[0].size = MAX(min_size, alloc->max_chunk_size);171    if (alloc->n_chunks == GGML_VBUFFER_MAX_CHUNKS - 1) {172        chunk->free_blocks[0].size = SIZE_MAX/2;173    }174    alloc->chunks[alloc->n_chunks] = chunk;175    alloc->n_chunks++;176    return alloc->n_chunks - 1;177}178 179#ifdef GGML_ALLOCATOR_DEBUG180static void add_allocated_tensor(struct ggml_dyn_tallocr * alloc, struct buffer_address addr, const struct ggml_tensor * tensor) {181    for (int i = 0; i < 1024; i++) {182        if (alloc->allocated_tensors[i].tensor == NULL) {183            alloc->allocated_tensors[i].tensor = tensor;184            alloc->allocated_tensors[i].addr = addr;185            return;186        }187    }188    GGML_ABORT("out of allocated_tensors");189}190static void remove_allocated_tensor(struct ggml_dyn_tallocr * alloc, struct buffer_address addr, const struct ggml_tensor * tensor) {191    for (int i = 0; i < 1024; i++) {192        if (alloc->allocated_tensors[i].addr.chunk == addr.chunk && alloc->allocated_tensors[i].addr.offset == addr.offset) {193            alloc->allocated_tensors[i].tensor = NULL;194            return;195        }196    }197    GGML_ABORT("tried to free tensor %s not found\n", tensor->name);198}199#endif200 201static struct buffer_address ggml_dyn_tallocr_alloc(struct ggml_dyn_tallocr * alloc, size_t size, const struct ggml_tensor * tensor) {202    size = aligned_offset(NULL, size, alloc->alignment);203 204    AT_PRINTF("%s: allocating %s (%zu bytes) - ", __func__, tensor->name, size);205 206    int best_fit_chunk = -1;207    int best_fit_block = -1;208    size_t max_avail = 0;209 210    // find the best fitting free block besides the last block, within any chunk211    for (int c = 0; c < alloc->n_chunks; ++c) {212        struct tallocr_chunk * chunk = alloc->chunks[c];213        size_t best_fit_size = SIZE_MAX;214        for (int i = 0; i < chunk->n_free_blocks - 1; i++) {215            struct free_block * block = &chunk->free_blocks[i];216            max_avail = MAX(max_avail, block->size);217            if (block->size >= size && block->size <= best_fit_size) {218                best_fit_chunk = c;219                best_fit_block = i;220                best_fit_size = block->size;221            }222        }223    }224 225    if (best_fit_block == -1) {226        // no suitable block found, try the last block (this may grow a chunks size)227        int64_t best_reuse = INT64_MIN;228        for (int c = 0; c < alloc->n_chunks; ++c) {229            struct tallocr_chunk * chunk = alloc->chunks[c];230            if (chunk->n_free_blocks > 0) {231                struct free_block * block = &chunk->free_blocks[chunk->n_free_blocks - 1];232                max_avail = MAX(max_avail, block->size);233                int64_t reuse_factor = chunk->max_size - block->offset - size;234                // reuse_factor < 0 : amount of extra memory that needs to be allocated235                // reuse_factor = 0 : allocated free space exactly matches tensor size236                // reuse_factor > 0 : superfluous memory that will remain unused237                bool better_reuse = best_reuse < 0 && reuse_factor > best_reuse;238                bool better_fit = reuse_factor >= 0 && reuse_factor < best_reuse;239                if (block->size >= size && (better_reuse || better_fit)) {240                    best_fit_chunk = c;241                    best_fit_block = chunk->n_free_blocks - 1;242                    best_reuse = reuse_factor;243                }244            }245        }246    }247 248    if (best_fit_block == -1) {249        // none of the existing chunks have enough space left250        best_fit_chunk = ggml_dyn_tallocr_new_chunk(alloc, size);251        best_fit_block = 0;252    }253    if (best_fit_chunk == -1) {254        // since the last chunk always has virtually endless memory, this should never happen255        GGML_LOG_ERROR("%s: not enough space in the buffer to allocate %zu bytes, largest block available %zu bytes\n",256            __func__, size, max_avail);257        GGML_ABORT("graph allocation: failed to reserve memory");258    }259 260    struct tallocr_chunk * chunk = alloc->chunks[best_fit_chunk];261    struct free_block    * block = &chunk->free_blocks[best_fit_block];262    struct buffer_address  addr  = {.chunk = best_fit_chunk, .offset = block->offset };263    block->offset += size;264    block->size -= size;265    if (block->size == 0) {266        // remove block if empty267        ggml_dyn_tallocr_remove_block(chunk, best_fit_block);268    }269 270    AT_PRINTF("block %d, offset %zu, chunk %d\n", best_fit_block, addr.offset, addr.chunk);271 272#ifdef GGML_ALLOCATOR_DEBUG273    add_allocated_tensor(alloc, addr, tensor);274    size_t cur_max = addr.offset + size;275    if (cur_max > chunk->max_size) {276        // sort allocated_tensors by chunk/offset277        for (int i = 0; i < 1024; i++) {278            for (int j = i + 1; j < 1024; j++) {279                if (ggml_buffer_address_less(alloc->allocated_tensors[j].addr, alloc->allocated_tensors[i].addr)) {280                    const struct ggml_tensor * tmp_tensor = alloc->allocated_tensors[i].tensor;281                    struct buffer_address tmp_addr = alloc->allocated_tensors[i].addr;282                    alloc->allocated_tensors[i].tensor = alloc->allocated_tensors[j].tensor;283                    alloc->allocated_tensors[i].addr = alloc->allocated_tensors[j].addr;284                    alloc->allocated_tensors[j].tensor = tmp_tensor;285                    alloc->allocated_tensors[j].addr = tmp_addr;286                }287            }288        }289        GGML_LOG_DEBUG("max_size[%d] = %.2f MB: tensors: ", addr.chunk, cur_max / 1024.0 / 1024.0);290        for (int i = 0; i < 1024; i++) {291            if (alloc->allocated_tensors[i].tensor) {292                GGML_LOG_DEBUG("%s [%d: %zx-%zx] (%.2f MB) ", alloc->allocated_tensors[i].tensor->name,293                    alloc->allocated_tensors[i].addr.chunk,294                    alloc->allocated_tensors[i].addr.offset,295                    alloc->allocated_tensors[i].addr.offset + ggml_nbytes(alloc->allocated_tensors[i].tensor),296                    ggml_nbytes(alloc->allocated_tensors[i].tensor) / 1024.0 / 1024.0);297            }298        }299        GGML_LOG_DEBUG("\n");300    }301#endif302 303    chunk->max_size = MAX(chunk->max_size, addr.offset + size);304 305    return addr;306 307    GGML_UNUSED(tensor);308}309 310// this is a very naive implementation, but for our case the number of free blocks should be very small311static void ggml_dyn_tallocr_free_bytes(struct ggml_dyn_tallocr * alloc, struct buffer_address addr, size_t size) {312    size = aligned_offset(NULL, size, alloc->alignment);313 314    struct tallocr_chunk * chunk = alloc->chunks[addr.chunk];315 316    // see if we can merge with an existing block317    for (int i = 0; i < chunk->n_free_blocks; i++) {318        struct free_block * block = &chunk->free_blocks[i];319        // check if ptr is at the end of the block320        if (block->offset + block->size == addr.offset) {321            block->size += size;322            // check if we can merge with the next block323            if (i < chunk->n_free_blocks - 1) {324                struct free_block * next = &chunk->free_blocks[i+1];325                if (block->offset + block->size == next->offset) {326                    block->size += next->size;327                    ggml_dyn_tallocr_remove_block(chunk, i+1);328                }329            }330            return;331        }332        // check if ptr is at the beginning of the block333        if (addr.offset + size == block->offset) {334            block->offset = addr.offset;335            block->size += size;336            // check if we can merge with the previous block337            if (i > 0) {338                struct free_block * prev = &chunk->free_blocks[i-1];339                if (prev->offset + prev->size == block->offset) {340                    prev->size += block->size;341                    ggml_dyn_tallocr_remove_block(chunk, i);342                }343            }344            return;345        }346    }347    // otherwise, add a new block348    ggml_dyn_tallocr_insert_block(chunk, addr.offset, size);349}350 351static void ggml_dyn_tallocr_reset(struct ggml_dyn_tallocr * alloc) {352    for (int i = 0; i < GGML_VBUFFER_MAX_CHUNKS; i++) {353        free(alloc->chunks[i]);354        alloc->chunks[i] = NULL;355    }356    alloc->n_chunks = 0;357 358#ifdef GGML_ALLOCATOR_DEBUG359    for (int i = 0; i < 1024; i++) {360        alloc->allocated_tensors[i].tensor = NULL;361    }362#endif363}364 365static struct ggml_dyn_tallocr * ggml_dyn_tallocr_new(size_t alignment, size_t max_buffer_size) {366    struct ggml_dyn_tallocr * alloc = (struct ggml_dyn_tallocr *)malloc(sizeof(struct ggml_dyn_tallocr));367 368    *alloc = (struct ggml_dyn_tallocr) {369        /*.alignment      = */ alignment,370        /*.max_chunk_size = */ MIN(max_buffer_size, SIZE_MAX/2), // clamp to avoid overflows371        /*.chunks         = */ {NULL},372        /*.n_chunks       = */ 0,373#ifdef GGML_ALLOCATOR_DEBUG374        /*.allocated_tensors = */ {{0}},375#endif376    };377 378    ggml_dyn_tallocr_reset(alloc);379 380    return alloc;381}382 383static void ggml_dyn_tallocr_free(struct ggml_dyn_tallocr * alloc) {384    for (int i = 0; i < alloc->n_chunks; ++i) {385        free(alloc->chunks[i]);386    }387    free(alloc);388}389 390static size_t ggml_dyn_tallocr_max_size(struct ggml_dyn_tallocr * alloc, int chunk) {391    return chunk < alloc->n_chunks ? alloc->chunks[chunk]->max_size : 0;392}393 394 395// virtual buffer with contiguous memory range, split into multiple backend buffers (chunks)396 397struct vbuffer {398    ggml_backend_buffer_t chunks[GGML_VBUFFER_MAX_CHUNKS];399};400 401static void ggml_vbuffer_free(struct vbuffer * buf) {402    if (buf == NULL) {403        return;404    }405    for (int i = 0; i < GGML_VBUFFER_MAX_CHUNKS; ++i) {406        ggml_backend_buffer_free(buf->chunks[i]);407    }408    free(buf);409}410 411static size_t ggml_vbuffer_chunk_size(struct vbuffer * buf, int chunk) {412    return buf->chunks[chunk] ? ggml_backend_buffer_get_size(buf->chunks[chunk]) : 0;413}414 415static size_t ggml_vbuffer_size(struct vbuffer * buf) {416    size_t size = 0;417    for (int i = 0; i < GGML_VBUFFER_MAX_CHUNKS && buf->chunks[i]; ++i) {418        size += ggml_backend_buffer_get_size(buf->chunks[i]);419    }420    return size;421}422 423static struct vbuffer * ggml_vbuffer_alloc(ggml_backend_buffer_type_t buft, const struct ggml_dyn_tallocr * talloc, enum ggml_backend_buffer_usage usage) {424    struct vbuffer * buf = (struct vbuffer *)calloc(1, sizeof(struct vbuffer));425    if (buf == NULL) {426        return NULL;427    }428 429    for (int n = 0; n < talloc->n_chunks; n++) {430        size_t chunk_size = talloc->chunks[n]->max_size;431        buf->chunks[n] = ggml_backend_buft_alloc_buffer(buft, chunk_size);432        if (buf->chunks[n] == NULL) {433            ggml_vbuffer_free(buf);434            return NULL;435        }436        ggml_backend_buffer_set_usage(buf->chunks[n], usage);437    }438    return buf;439}440 441static void ggml_vbuffer_tensor_alloc(struct vbuffer * buf, struct ggml_tensor * tensor, struct buffer_address buf_addr) {442    void * base = ggml_backend_buffer_get_base(buf->chunks[buf_addr.chunk]);443    void * addr = (char *)base + buf_addr.offset;444    ggml_backend_tensor_alloc(buf->chunks[buf_addr.chunk], tensor, addr);445}446 447static void ggml_vbuffer_reset(struct vbuffer * buf) {448    for (int i = 0; i < GGML_VBUFFER_MAX_CHUNKS && buf->chunks[i]; ++i) {449        ggml_backend_buffer_reset(buf->chunks[i]);450    }451}452 453 454/////////////////////////////////////455 456// graph allocator457 458struct hash_node {459    int n_children;460    int n_views;461    int buffer_id;462    struct buffer_address addr;463    bool allocated;464};465 466struct tensor_alloc {467    int buffer_id;468    struct buffer_address addr;469    size_t size_max; // 0 = pre-allocated, unused, or view470};471 472struct leaf_alloc {473    struct tensor_alloc leaf;474};475 476struct node_alloc {477    struct tensor_alloc dst;478    struct tensor_alloc src[GGML_MAX_SRC];479};480 481struct ggml_gallocr {482    ggml_backend_buffer_type_t * bufts; // [n_buffers]483    struct vbuffer ** buffers; // [n_buffers]484    struct ggml_dyn_tallocr ** buf_tallocs; // [n_buffers]485    int n_buffers;486 487    struct ggml_hash_set hash_set;488    struct hash_node * hash_values; // [hash_set.size]489 490    struct node_alloc * node_allocs; // [n_nodes]491    int n_nodes;492 493    struct leaf_alloc * leaf_allocs; // [n_leafs]494    int n_leafs;495};496 497ggml_gallocr_t ggml_gallocr_new_n(ggml_backend_buffer_type_t * bufts, int n_bufs) {498    ggml_gallocr_t galloc = (ggml_gallocr_t)calloc(1, sizeof(struct ggml_gallocr));499    GGML_ASSERT(galloc != NULL);500 501    galloc->bufts = calloc(n_bufs, sizeof(ggml_backend_buffer_type_t));502    GGML_ASSERT(galloc->bufts != NULL);503 504    galloc->buffers = calloc(n_bufs, sizeof(struct vbuffer *));505    GGML_ASSERT(galloc->buffers != NULL);506 507    galloc->buf_tallocs = calloc(n_bufs, sizeof(struct ggml_dyn_tallocr *));508    GGML_ASSERT(galloc->buf_tallocs != NULL);509 510    for (int i = 0; i < n_bufs; i++) {511        galloc->bufts[i] = bufts[i];512        galloc->buffers[i] = NULL;513 514        // check if the same buffer type is used multiple times and reuse the same allocator515        for (int j = 0; j < i; j++) {516            if (bufts[i] == bufts[j]) {517                galloc->buf_tallocs[i] = galloc->buf_tallocs[j];518                break;519            }520        }521 522        if (galloc->buf_tallocs[i] == NULL) {523            size_t alignment = ggml_backend_buft_get_alignment(bufts[i]);524            size_t max_size = ggml_backend_buft_get_max_size(bufts[i]);525            galloc->buf_tallocs[i] = ggml_dyn_tallocr_new(alignment, max_size);526        }527    }528    galloc->n_buffers = n_bufs;529 530    return galloc;531}532 533ggml_gallocr_t ggml_gallocr_new(ggml_backend_buffer_type_t buft) {534    return ggml_gallocr_new_n(&buft, 1);535}536 537void ggml_gallocr_free(ggml_gallocr_t galloc) {538    if (galloc == NULL) {539        return;540    }541 542    for (int i = 0; i < galloc->n_buffers; i++) {543        if (galloc->buffers != NULL) {544            // skip if already freed545            bool freed = false;546            for (int j = 0; j < i; j++) {547                if (galloc->buffers[j] == galloc->buffers[i]) {548                    freed = true;549                    break;550                }551            }552            if (!freed) {553                ggml_vbuffer_free(galloc->buffers[i]);554            }555        }556        if (galloc->buf_tallocs != NULL) {557            // skip if already freed558            bool freed = false;559            for (int j = 0; j < i; j++) {560                if (galloc->buf_tallocs[j] == galloc->buf_tallocs[i]) {561                    freed = true;562                    break;563                }564            }565            if (!freed) {566                ggml_dyn_tallocr_free(galloc->buf_tallocs[i]);567            }568        }569    }570 571    ggml_hash_set_free(&galloc->hash_set);572    free(galloc->hash_values);573    free(galloc->bufts);574    free(galloc->buffers);575    free(galloc->buf_tallocs);576    free(galloc->node_allocs);577    free(galloc->leaf_allocs);578    free(galloc);579}580 581typedef struct ggml_gallocr * ggml_gallocr_t;582 583static struct hash_node * ggml_gallocr_hash_get(ggml_gallocr_t galloc, struct ggml_tensor * t) {584    size_t i = ggml_hash_find_or_insert(&galloc->hash_set, t);585    return &galloc->hash_values[i];586}587 588static bool ggml_gallocr_is_own(ggml_gallocr_t galloc, struct ggml_tensor * t) {589    return ggml_gallocr_hash_get(galloc, t)->allocated;590}591 592static bool ggml_gallocr_is_allocated(ggml_gallocr_t galloc, struct ggml_tensor * t) {593    return t->data != NULL // tensor data already set externally594        || t->buffer // tensor on external buffer (but not yet allocated)595        || ggml_gallocr_is_own(galloc, t); // tensor will be allocated by galloc596}597 598// free the extra space at the end if the new tensor is smaller599static void ggml_gallocr_free_extra_space(ggml_gallocr_t galloc, struct ggml_tensor * node, struct ggml_tensor * parent) {600    struct hash_node * hn = ggml_gallocr_hash_get(galloc, node);601    struct hash_node * p_hn = ggml_gallocr_hash_get(galloc, parent);602 603    size_t parent_size = ggml_backend_buft_get_alloc_size(galloc->bufts[p_hn->buffer_id], parent);604    size_t node_size = ggml_backend_buft_get_alloc_size(galloc->bufts[hn->buffer_id], node);605 606    GGML_ASSERT(parent_size >= node_size);607 608    // note: we want after the freeing the chunks to continue to be aligned609    struct ggml_dyn_tallocr * p_alloc = galloc->buf_tallocs[p_hn->buffer_id];610    parent_size = aligned_offset(NULL, parent_size, p_alloc->alignment);611    node_size = aligned_offset(NULL, node_size, p_alloc->alignment);612 613    if (parent_size > node_size) {614        struct buffer_address p_addr = p_hn->addr;615        p_addr.offset += node_size;616        size_t extra_size = parent_size - node_size;617        AT_PRINTF("freeing extra %zu bytes from parent %s for %s\n", extra_size, parent->name, node->name);618        ggml_dyn_tallocr_free_bytes(p_alloc, p_addr, extra_size);619    }620}621 622static void ggml_gallocr_allocate_node(ggml_gallocr_t galloc, struct ggml_tensor * node, int buffer_id) {623    GGML_ASSERT(buffer_id >= 0);624    struct hash_node * hn = ggml_gallocr_hash_get(galloc, node);625 626    if (!ggml_gallocr_is_allocated(galloc, node) && !ggml_impl_is_view(node)) {627        hn->allocated = true;628        assert(hn->addr.offset == 0);629 630        // try to reuse a parent's buffer (inplace)631        if (ggml_op_can_inplace(node->op)) {632            for (int i = 0; i < GGML_MAX_SRC; i++) {633                struct ggml_tensor * parent = node->src[i];634                if (parent == NULL) {635                    continue;636                }637 638                // if the node's data is external, then we cannot re-use it639                if (!ggml_gallocr_is_own(galloc, parent)) {640                    AT_PRINTF("not reusing parent %s for %s as %p is external\n", parent->name, node->name, parent->data);641                    continue;642                }643 644                // outputs cannot be reused645                if (parent->flags & GGML_TENSOR_FLAG_OUTPUT || (parent->view_src != NULL && parent->view_src->flags & GGML_TENSOR_FLAG_OUTPUT)) {646                    AT_PRINTF("not reusing parent %s for %s as it is an output\n", parent->name, node->name);647                    continue;648                }649 650                if (!ggml_are_same_layout(node, parent)) {651                    AT_PRINTF("not reusing parent %s for %s as layouts are different\n", parent->name, node->name);652                    continue;653                }654 655                struct hash_node * p_hn = ggml_gallocr_hash_get(galloc, parent);656                if (p_hn->n_children == 1 && p_hn->n_views == 0) {657                    if (ggml_impl_is_view(parent)) {658                        struct ggml_tensor * view_src = parent->view_src;659                        struct hash_node * view_src_hn = ggml_gallocr_hash_get(galloc, view_src);660                        if (view_src_hn->n_views == 1 && view_src_hn->n_children == 0 && view_src->data == parent->data) {661                            AT_PRINTF("reusing view parent %s (%s) for %s\n", parent->name, view_src->name, node->name);662                            assert(view_src_hn->addr.chunk == p_hn->addr.chunk && view_src_hn->addr.offset == p_hn->addr.offset);663                            hn->buffer_id = p_hn->buffer_id;664                            hn->addr = p_hn->addr;665                            p_hn->allocated = false; // avoid freeing the parent666                            view_src_hn->allocated = false;667                            ggml_gallocr_free_extra_space(galloc, node, view_src);668                            return;669                        }670                    } else {671                        AT_PRINTF("reusing parent %s for %s\n", parent->name, node->name);672                        hn->buffer_id = p_hn->buffer_id;673                        hn->addr = p_hn->addr;674                        p_hn->allocated = false; // avoid freeing the parent675                        ggml_gallocr_free_extra_space(galloc, node, parent);676                        return;677                    }678                }679            }680        }681        // allocate tensor from the buffer682        struct ggml_dyn_tallocr * alloc = galloc->buf_tallocs[buffer_id];683        ggml_backend_buffer_type_t buft = galloc->bufts[buffer_id];684        size_t size = ggml_backend_buft_get_alloc_size(buft, node);685        hn->buffer_id = buffer_id;686        hn->addr = ggml_dyn_tallocr_alloc(alloc, size, node);687    }688}689 690static void ggml_gallocr_free_node(ggml_gallocr_t galloc, struct ggml_tensor * node) {691    // graph outputs are never freed692    if (node->flags & GGML_TENSOR_FLAG_OUTPUT) {693        AT_PRINTF("not freeing output %s\n", node->name);694        return;695    }696 697    struct hash_node * hn = ggml_gallocr_hash_get(galloc, node);698    int buffer_id = hn->buffer_id;699    struct ggml_dyn_tallocr * alloc = galloc->buf_tallocs[buffer_id];700    ggml_backend_buffer_type_t buft = galloc->bufts[buffer_id];701    size_t size = ggml_backend_buft_get_alloc_size(buft, node);702 703    AT_PRINTF("%s: freeing %s at {chunk=%d, offset=%zu} (%zu bytes) - n_free_blocks = %d\n",704        __func__, node->name, hn->addr.chunk, hn->addr.offset, size, alloc->chunks[hn->addr.chunk]->n_free_blocks);705#ifdef GGML_ALLOCATOR_DEBUG706    remove_allocated_tensor(alloc, hn->addr, node);707#endif708 709    ggml_dyn_tallocr_free_bytes(alloc, hn->addr, size);710    hn->allocated = false;711}712 713static int get_node_buffer_id(const int * node_buffer_ids, int i) {714    return node_buffer_ids ? node_buffer_ids[i] : 0;715}716 717static void ggml_gallocr_alloc_graph_impl(ggml_gallocr_t galloc, struct ggml_cgraph * graph, const int * node_buffer_ids, const int * leaf_buffer_ids) {718    // clear hash tables719    ggml_hash_set_reset(&galloc->hash_set);720    memset(galloc->hash_values, 0, sizeof(struct hash_node) * galloc->hash_set.size);721 722    // allocate leafs723    // these may be tensors that the application is not using in the graph, but may still want to allocate for other purposes724    for (int i = 0; i < graph->n_leafs; i++) {725        struct ggml_tensor * leaf = graph->leafs[i];726        ggml_gallocr_allocate_node(galloc, leaf, get_node_buffer_id(leaf_buffer_ids, i));727    }728 729    // count number of children and views730    // allocate other graph inputs and leafs first to avoid overwriting them731    for (int i = 0; i < graph->n_nodes; i++) {732        struct ggml_tensor * node = graph->nodes[i];733 734        // TODO: better way to add external dependencies735        // GGML_OP_NONE does not appear normally in the graph nodes, but is used by ggml-backend to add dependencies to736        // control when some tensors are allocated and freed. in this case, the dependencies are in `src`, but the node737        // itself is never used and should not be considered a dependency738        if (ggml_impl_is_view(node) && node->op != GGML_OP_NONE) {739            struct ggml_tensor * view_src = node->view_src;740            ggml_gallocr_hash_get(galloc, view_src)->n_views += 1;741        }742 743        if (node->flags & GGML_TENSOR_FLAG_INPUT) {744            ggml_gallocr_allocate_node(galloc, graph->nodes[i], get_node_buffer_id(node_buffer_ids, i));745        }746 747        for (int j = 0; j < GGML_MAX_SRC; j++) {748            struct ggml_tensor * src = node->src[j];749            if (src == NULL) {750                continue;751            }752 753            ggml_gallocr_hash_get(galloc, src)->n_children += 1;754 755            // allocate explicit inputs756            if (src->flags & GGML_TENSOR_FLAG_INPUT) {757                ggml_gallocr_allocate_node(galloc, src, get_node_buffer_id(node_buffer_ids, i));758            }759        }760    }761 762    // allocate tensors763    for (int i = 0; i < graph->n_nodes; i++) {764        struct ggml_tensor * node = graph->nodes[i];765        int buffer_id = get_node_buffer_id(node_buffer_ids, i);766 767        // allocate parents (only leafs need to be allocated at this point)768        for (int j = 0; j < GGML_MAX_SRC; j++) {769            struct ggml_tensor * parent = node->src[j];770            if (parent == NULL) {771                continue;772            }773            ggml_gallocr_allocate_node(galloc, parent, buffer_id);774        }775 776        // allocate node777        ggml_gallocr_allocate_node(galloc, node, buffer_id);778 779        AT_PRINTF("exec: %s (%s) <= ", ggml_op_desc(node), node->name);780        for (int j = 0; j < GGML_MAX_SRC; j++) {781            struct ggml_tensor * parent = node->src[j];782            if (parent == NULL) {783                continue;784            }785            AT_PRINTF("%s", parent->name);786            if (j < GGML_MAX_SRC - 1 && node->src[j + 1] != NULL) {787                AT_PRINTF(", ");788            }789        }790        AT_PRINTF("\n");791 792        // update parents793        for (int j = 0; j < GGML_MAX_SRC; j++) {794            struct ggml_tensor * parent = node->src[j];795            if (parent == NULL) {796                continue;797            }798            struct hash_node * p_hn = ggml_gallocr_hash_get(galloc, parent);799            p_hn->n_children -= 1;800 801            AT_PRINTF("parent %s: %d children, %d views, allocated: %d\n",802                parent->name, p_hn->n_children, p_hn->n_views, p_hn->allocated);803 804            if (p_hn->n_children == 0 && p_hn->n_views == 0) {805                if (ggml_impl_is_view(parent)) {806                    struct ggml_tensor * view_src = parent->view_src;807                    struct hash_node * view_src_hn = ggml_gallocr_hash_get(galloc, view_src);808                    view_src_hn->n_views -= 1;809                    AT_PRINTF("view_src %s: %d children, %d views\n",810                        view_src->name, view_src_hn->n_children, view_src_hn->n_views);811                    if (view_src_hn->n_views == 0 && view_src_hn->n_children == 0 && view_src_hn->allocated) {812                        ggml_gallocr_free_node(galloc, view_src);813                    }814                }815                else if (p_hn->allocated) {816                    ggml_gallocr_free_node(galloc, parent);817                }818            }819            AT_PRINTF("\n");820        }821    }822}823 824static bool ggml_gallocr_reserve_n_impl(825        ggml_gallocr_t galloc, struct ggml_cgraph * graph, const int * node_buffer_ids, const int * leaf_buffer_ids, bool no_alloc) {826    size_t min_hash_size = graph->n_nodes + graph->n_leafs;827    // add 25% margin to avoid hash collisions828    min_hash_size += min_hash_size / 4;829 830    // initialize hash table831    if (galloc->hash_set.size < min_hash_size) {832        ggml_hash_set_free(&galloc->hash_set);833        galloc->hash_set = ggml_hash_set_new(min_hash_size);834        GGML_ASSERT(galloc->hash_set.keys != NULL);835 836        free(galloc->hash_values);837        galloc->hash_values = malloc(sizeof(struct hash_node) * galloc->hash_set.size);838        GGML_ASSERT(galloc->hash_values != NULL);839    }840 841    // reset allocators842    for (int i = 0; i < galloc->n_buffers; i++) {843        ggml_dyn_tallocr_reset(galloc->buf_tallocs[i]);844    }845 846    // allocate in hash table847    ggml_gallocr_alloc_graph_impl(galloc, graph, node_buffer_ids, leaf_buffer_ids);848 849    // set the node_allocs from the hash table850    if (galloc->n_nodes < graph->n_nodes) {851        free(galloc->node_allocs);852        galloc->node_allocs = calloc(graph->n_nodes, sizeof(struct node_alloc));853        GGML_ASSERT(galloc->node_allocs != NULL);854    }855    galloc->n_nodes = graph->n_nodes;856    for (int i = 0; i < graph->n_nodes; i++) {857        struct ggml_tensor * node = graph->nodes[i];858        struct node_alloc * node_alloc = &galloc->node_allocs[i];859        if (node->view_src || node->data) {860            node_alloc->dst.buffer_id = -1;861            node_alloc->dst.addr = GGML_BUFFER_ADDRESS_INVALID;862            node_alloc->dst.size_max = 0;863        } else {864            struct hash_node * hn = ggml_gallocr_hash_get(galloc, node);865            node_alloc->dst.buffer_id = hn->buffer_id;866            node_alloc->dst.addr = hn->addr;867            node_alloc->dst.size_max  = ggml_backend_buft_get_alloc_size(galloc->bufts[hn->buffer_id], node);868        }869        for (int j = 0; j < GGML_MAX_SRC; j++) {870            struct ggml_tensor * src = node->src[j];871            if (!src || src->view_src || src->data) {872                node_alloc->src[j].buffer_id = -1;873                node_alloc->src[j].addr = GGML_BUFFER_ADDRESS_INVALID;874                node_alloc->src[j].size_max = 0;875            } else {876                struct hash_node * hn = ggml_gallocr_hash_get(galloc, src);877                node_alloc->src[j].buffer_id = hn->buffer_id;878                node_alloc->src[j].addr = hn->addr;879                node_alloc->src[j].size_max = ggml_backend_buft_get_alloc_size(galloc->bufts[hn->buffer_id], src);880            }881        }882    }883    if (galloc->n_leafs < graph->n_leafs) {884        free(galloc->leaf_allocs);885        galloc->leaf_allocs = calloc(graph->n_leafs, sizeof(galloc->leaf_allocs[0]));886        GGML_ASSERT(galloc->leaf_allocs != NULL);887    }888    galloc->n_leafs = graph->n_leafs;889    for (int i = 0; i < graph->n_leafs; i++) {890        struct ggml_tensor * leaf = graph->leafs[i];891        struct hash_node * hn = ggml_gallocr_hash_get(galloc, leaf);892        if (leaf->view_src || leaf->data) {893            galloc->leaf_allocs[i].leaf.buffer_id = -1;894            galloc->leaf_allocs[i].leaf.addr = GGML_BUFFER_ADDRESS_INVALID;895            galloc->leaf_allocs[i].leaf.size_max = 0;896        } else {897            galloc->leaf_allocs[i].leaf.buffer_id = hn->buffer_id;898            galloc->leaf_allocs[i].leaf.addr = hn->addr;899            galloc->leaf_allocs[i].leaf.size_max = ggml_backend_buft_get_alloc_size(galloc->bufts[hn->buffer_id], leaf);900        }901    }902 903    // reallocate buffers if needed904    for (int i = 0; i < galloc->n_buffers; i++) {905        // if the buffer type is used multiple times, we reuse the same buffer906        for (int j = 0; j < i; j++) {907            if (galloc->buf_tallocs[j] == galloc->buf_tallocs[i]) {908                galloc->buffers[i] = galloc->buffers[j];909                break;910            }911        }912 913        // even if there are no tensors allocated in this buffer, we still need to allocate it to initialize views914        bool realloc = galloc->buffers[i] == NULL;915        size_t new_size = 0;916        for (int c = 0; c < galloc->buf_tallocs[i]->n_chunks; c++) {917            size_t cur_chunk_size = galloc->buffers[i] ? ggml_vbuffer_chunk_size(galloc->buffers[i], c) : 0;918            size_t new_chunk_size = ggml_dyn_tallocr_max_size(galloc->buf_tallocs[i], c);919            new_size += new_chunk_size;920            if (new_chunk_size > cur_chunk_size) {921                realloc = true;922            }923        }924        if (realloc) {925#ifndef NDEBUG926            {927                size_t cur_size = galloc->buffers[i] ? ggml_vbuffer_size(galloc->buffers[i]) : 0;928                if (cur_size > 0) {929                    GGML_LOG_DEBUG("%s: reallocating %s buffer from size %.02f MiB to %.02f MiB\n",930                        __func__, ggml_backend_buft_name(galloc->bufts[i]), cur_size / 1024.0 / 1024.0, new_size / 1024.0 / 1024.0);931                }932            }933#endif934            ggml_vbuffer_free(galloc->buffers[i]);935            if (no_alloc) {936                galloc->buffers[i] = NULL;937            } else {938                galloc->buffers[i] = ggml_vbuffer_alloc(galloc->bufts[i], galloc->buf_tallocs[i], GGML_BACKEND_BUFFER_USAGE_COMPUTE);939                if (galloc->buffers[i] == NULL) {940                    GGML_LOG_ERROR("%s: failed to allocate %s buffer of size %zu\n", __func__, ggml_backend_buft_name(galloc->bufts[i]), new_size);941                    return false;942                }943            }944        }945    }946 947    return true;948}949 950void ggml_gallocr_reserve_n_size(951        ggml_gallocr_t galloc, struct ggml_cgraph * graph, const int * node_buffer_ids, const int * leaf_buffer_ids, size_t * sizes) {952    GGML_ASSERT(ggml_gallocr_reserve_n_impl(galloc, graph, node_buffer_ids, leaf_buffer_ids, /*no_alloc =*/ true));953    for (int i = 0; i < galloc->n_buffers; i++) {954        sizes[i] = 0;955        for (int c = 0; c < galloc->buf_tallocs[i]->n_chunks; c++) {956            sizes[i] += galloc->buf_tallocs[i]->chunks[c]->max_size;957        }958    }959}960 961bool ggml_gallocr_reserve_n(ggml_gallocr_t galloc, struct ggml_cgraph * graph, const int * node_buffer_ids, const int * leaf_buffer_ids) {962    return ggml_gallocr_reserve_n_impl(galloc, graph, node_buffer_ids, leaf_buffer_ids, /*no_alloc =*/ false);963}964 965bool ggml_gallocr_reserve(ggml_gallocr_t galloc, struct ggml_cgraph *graph) {966    return ggml_gallocr_reserve_n(galloc, graph, NULL, NULL);967}968 969static void ggml_gallocr_init_tensor(ggml_gallocr_t galloc, struct ggml_tensor * tensor, struct tensor_alloc * tensor_alloc) {970    int buffer_id = tensor_alloc->buffer_id;971    assert(tensor->data || tensor->view_src || ggml_backend_buft_get_alloc_size(galloc->bufts[buffer_id], tensor) <= tensor_alloc->size_max);972 973    if (tensor->view_src != NULL) {974        if (tensor->buffer == NULL) {975            assert(tensor_alloc->addr.offset == SIZE_MAX);976            if (tensor->view_src->buffer == NULL) {977                // this tensor was allocated without ggml-backend978                return;979            }980            ggml_backend_view_init(tensor);981        }982    } else {983        if (tensor->data == NULL) {984            assert(tensor_alloc->addr.offset != SIZE_MAX);985            assert(ggml_backend_buft_get_alloc_size(galloc->bufts[buffer_id], tensor) <= tensor_alloc->size_max);986            ggml_vbuffer_tensor_alloc(galloc->buffers[buffer_id], tensor, tensor_alloc->addr);987        } else {988            if (tensor->buffer == NULL) {989                // this tensor was allocated without ggml-backend990                return;991            }992        }993    }994}995 996static bool ggml_gallocr_node_needs_realloc(ggml_gallocr_t galloc, struct ggml_tensor * node, struct tensor_alloc * talloc) {997    size_t node_size = 0;998    if (!node->data && !node->view_src) {999        // If we previously had data but don't now then reallocate1000        if (talloc->buffer_id < 0) {1001            return false;1002        }1003        node_size = ggml_backend_buft_get_alloc_size(galloc->bufts[talloc->buffer_id], node);1004    }1005    return talloc->size_max >= node_size;1006}1007 1008static bool ggml_gallocr_needs_realloc(ggml_gallocr_t galloc, struct ggml_cgraph * graph) {1009    if (galloc->n_nodes != graph->n_nodes) {1010#ifndef NDEBUG1011        GGML_LOG_DEBUG("%s: graph has different number of nodes\n", __func__);1012#endif1013        return true;1014    }1015 1016    if (galloc->n_leafs != graph->n_leafs) {1017#ifndef NDEBUG1018        GGML_LOG_DEBUG("%s: graph has different number of leafs\n", __func__);1019#endif1020        return true;1021    }1022 1023    for (int i = 0; i < graph->n_nodes; i++) {1024        struct ggml_tensor * node = graph->nodes[i];1025        struct node_alloc * node_alloc = &galloc->node_allocs[i];1026 1027        if (!ggml_gallocr_node_needs_realloc(galloc, node, &node_alloc->dst)) {1028#ifndef NDEBUG1029            GGML_LOG_DEBUG("%s: node %s is not valid\n", __func__, node->name);1030#endif1031            return true;1032        }1033 1034        for (int j = 0; j < GGML_MAX_SRC; j++) {1035            struct ggml_tensor * src = node->src[j];1036            if (src == NULL) {1037                continue;1038            }1039            if (!ggml_gallocr_node_needs_realloc(galloc, src, &node_alloc->src[j])) {1040#ifndef NDEBUG1041                GGML_LOG_DEBUG("%s: src %d (%s) of node %s is not valid\n", __func__, j, src->name, node->name);1042#endif1043                return true;1044            }1045        }1046    }1047 1048    return false;1049}1050 1051bool ggml_gallocr_alloc_graph(ggml_gallocr_t galloc, struct ggml_cgraph * graph) {1052    if (ggml_gallocr_needs_realloc(galloc, graph)) {1053        if (galloc->n_buffers == 1) {1054#ifndef NDEBUG1055            GGML_LOG_DEBUG("%s: reallocating buffers automatically\n", __func__);1056#endif1057            if (!ggml_gallocr_reserve(galloc, graph)) {1058                return false;1059            }1060        } else {1061#ifndef NDEBUG1062            GGML_LOG_DEBUG("%s: cannot reallocate multi buffer graph automatically, call reserve\n", __func__);1063#endif1064            return false;1065        }1066    }1067 1068    // reset buffers1069    for (int i = 0; i < galloc->n_buffers; i++) {1070        if (galloc->buffers[i] != NULL) {1071            ggml_vbuffer_reset(galloc->buffers[i]);1072        }1073    }1074 1075    // allocate the graph tensors from the previous assignments1076    // leafs1077    for (int i = 0; i < graph->n_leafs; i++) {1078        struct ggml_tensor * leaf = graph->leafs[i];1079        struct leaf_alloc * leaf_alloc = &galloc->leaf_allocs[i];1080        ggml_gallocr_init_tensor(galloc, leaf, &leaf_alloc->leaf);1081    }1082    // nodes1083    for (int i = 0; i < graph->n_nodes; i++) {1084        struct ggml_tensor * node = graph->nodes[i];1085        struct node_alloc * node_alloc = &galloc->node_allocs[i];1086        for (int j = 0; j < GGML_MAX_SRC; j++) {1087            struct ggml_tensor * src = node->src[j];1088            if (src == NULL) {1089                continue;1090            }1091            ggml_gallocr_init_tensor(galloc, src, &node_alloc->src[j]);1092        }1093        ggml_gallocr_init_tensor(galloc, node, &node_alloc->dst);1094    }1095 1096    return true;1097}1098 1099size_t ggml_gallocr_get_buffer_size(ggml_gallocr_t galloc, int buffer_id) {1100    GGML_ASSERT(buffer_id >= 0 && buffer_id < galloc->n_buffers);1101 1102    if (galloc->buffers[buffer_id] == NULL) {1103        return 0;1104    }1105 1106    for (int i = 0; i < buffer_id; i++) {1107        if (galloc->buffers[i] == galloc->buffers[buffer_id]) {1108            // this buffer is the same as a previous one due to the same buffer type being used multiple times1109            // only return the buffer size the first time it appears to avoid double counting1110            return 0;1111        }1112    }1113 1114    return ggml_vbuffer_size(galloc->buffers[buffer_id]);1115}1116 1117// utils1118 1119static void free_buffers(ggml_backend_buffer_t ** buffers, const size_t * n_buffers) {1120    for (size_t i = 0; i < *n_buffers; i++) {1121        ggml_backend_buffer_free((*buffers)[i]);1122    }1123    free(*buffers);1124}1125 1126static bool alloc_tensor_range(struct ggml_context * ctx,1127        struct ggml_tensor * first, struct ggml_tensor * last,1128        ggml_backend_buffer_type_t buft, size_t size,1129        ggml_backend_buffer_t ** buffers, size_t * n_buffers) {1130 1131    ggml_backend_buffer_t buffer = ggml_backend_buft_alloc_buffer(buft, size);1132    if (buffer == NULL) {1133        GGML_LOG_ERROR("%s: failed to allocate %s buffer of size %zu\n", __func__, ggml_backend_buft_name(buft), size);1134        free_buffers(buffers, n_buffers);1135        return false;1136    }1137 1138    *buffers = realloc(*buffers, sizeof(ggml_backend_buffer_t) * (*n_buffers + 1));1139    (*buffers)[(*n_buffers)++] = buffer;1140 1141    struct ggml_tallocr tallocr = ggml_tallocr_new(buffer);1142 1143    for (struct ggml_tensor * t = first; t != last; t = ggml_get_next_tensor(ctx, t)) {1144        enum ggml_status status = GGML_STATUS_SUCCESS;1145        if (t->data == NULL) {1146            if (t->view_src == NULL) {1147                status = ggml_tallocr_alloc(&tallocr, t);1148            } else if (t->buffer == NULL) {1149                status = ggml_backend_view_init(t);1150            }1151        } else {1152            if (t->view_src != NULL && t->buffer == NULL) {1153                // view of a pre-allocated tensor1154                status = ggml_backend_view_init(t);1155            }1156        }1157        if (status != GGML_STATUS_SUCCESS) {1158            GGML_LOG_ERROR("%s: failed to initialize tensor %s\n", __func__, t->name);1159            free_buffers(buffers, n_buffers);1160            return false;1161        }1162    }1163 1164    return true;1165}1166 1167static ggml_backend_buffer_t ggml_backend_alloc_ctx_tensors_from_buft_impl(1168        struct ggml_context * ctx, ggml_backend_buffer_type_t buft, size_t * nbytes_total, bool no_alloc) {1169    GGML_ASSERT(ggml_get_no_alloc(ctx) == true);1170 1171    size_t alignment = ggml_backend_buft_get_alignment(buft);1172    size_t max_size = ggml_backend_buft_get_max_size(buft);1173 1174    ggml_backend_buffer_t * buffers = NULL;1175    size_t n_buffers = 0;1176    *nbytes_total = 0;1177 1178    size_t cur_buf_size = 0;1179    struct ggml_tensor * first = ggml_get_first_tensor(ctx);1180    for (struct ggml_tensor * t = first; t != NULL; t = ggml_get_next_tensor(ctx, t)) {1181        size_t this_size = 0;1182        if (t->data == NULL && t->view_src == NULL) {1183            this_size = GGML_PAD(ggml_backend_buft_get_alloc_size(buft, t), alignment);1184        }1185 1186        if (cur_buf_size > 0 && (cur_buf_size + this_size) > max_size) {1187            // allocate tensors in the current buffer1188            if (!no_alloc && !alloc_tensor_range(ctx, first, t, buft, cur_buf_size, &buffers, &n_buffers)) {1189                return NULL;1190            }1191            first = t;1192            *nbytes_total += cur_buf_size;1193            cur_buf_size = this_size;1194        } else {1195            cur_buf_size += this_size;1196        }1197    }1198 1199    // allocate remaining tensors1200    if (cur_buf_size > 0) {

Showing the first 1,200 of 1249 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai