Team Ai
Datasetpublic

echodict/llama.cpp

version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes479downloads
llama-ext.h91 linesDownload Raw Back to src
1#pragma once2 3// this is a staging header for new llama.cpp API4// breaking changes and C++ are allowed. everything here should be considered WIP5 6#include "llama.h"7 8#include <cstdint>9#include <map>10 11// Reserve a new compute graph. It is valid until the next call to llama_graph_reserve.12LLAMA_API struct ggml_cgraph * llama_graph_reserve(13        struct llama_context * ctx,14        uint32_t n_tokens,15        uint32_t n_seqs,16        uint32_t n_outputs);17 18// Get the default ggml_type for a given ftype.19LLAMA_API ggml_type llama_ftype_get_default_type(llama_ftype ftype);20 21struct quantize_state_impl;22 23LLAMA_API quantize_state_impl * llama_quant_init(24        const llama_model * model,25        const llama_model_quantize_params * params);26 27LLAMA_API void llama_quant_free(quantize_state_impl * qs);28 29// Descriptor for constructing a mock model for quantization testing.30struct llama_quant_model_desc {31    const char * architecture;32    uint32_t n_embd;33    uint32_t n_ff;34    uint32_t n_layer;35    uint32_t n_head;36    uint32_t n_head_kv;37    uint32_t n_expert;38    uint32_t n_embd_head_k;39    uint32_t n_embd_head_v;40};41 42// Create a mock model from a metadata descriptor (for testing).43// The returned model must be freed with llama_model_free().44LLAMA_API llama_model * llama_quant_model_from_metadata(const llama_quant_model_desc * desc);45 46// Returns true if this tensor should be quantized (based on name, dims, params).47LLAMA_API bool llama_quant_tensor_allows_quantization(48        const quantize_state_impl * qs,49        const ggml_tensor * tensor);50 51// Compute quantization type assignments for a list of tensors.52// All tensors should be quantizable (use llama_quant_tensor_allows_quantization to filter).53// result_types: caller-allocated array of n_tensors elements, filled with assigned types.54LLAMA_API void llama_quant_compute_types(55        quantize_state_impl * qs,56        llama_ftype ftype,57        ggml_tensor ** tensors,58        ggml_type * result_types,59        size_t n_tensors);60 61//62// device memory querying63//64 65// "memory" as in physical memory for a buffer type, in bytes66struct llama_memory_breakdown_data {67    size_t model   = 0; // memory allocated for the model68    size_t context = 0; // memory allocated for the context69    size_t compute = 0; // memory allocated for temporary compute buffers70 71    size_t total() const {72        return model + context + compute;73    }74};75 76struct llama_device_memory_data {77    int64_t total;78    int64_t free;79    llama_memory_breakdown_data mb;80};81 82// TODO: convert to C-style data structure83using llama_memory_breakdown = std::map<ggml_backend_buffer_type_t, llama_memory_breakdown_data>;84 85LLAMA_API int32_t llama_model_n_expert (const struct llama_model * model);86LLAMA_API int32_t llama_model_n_devices(const struct llama_model * model);87 88LLAMA_API ggml_backend_dev_t llama_model_get_device(const struct llama_model * model, int i);89 90LLAMA_API llama_memory_breakdown llama_get_memory_breakdown(const struct llama_context * ctx);91