echodict/llama.cpp
version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786
0479
1#pragma once2 3// this is a staging header for new llama.cpp API4// breaking changes and C++ are allowed. everything here should be considered WIP5 6#include "llama.h"7 8#include <cstdint>9#include <map>10 11// Reserve a new compute graph. It is valid until the next call to llama_graph_reserve.12LLAMA_API struct ggml_cgraph * llama_graph_reserve(13 struct llama_context * ctx,14 uint32_t n_tokens,15 uint32_t n_seqs,16 uint32_t n_outputs);17 18// Get the default ggml_type for a given ftype.19LLAMA_API ggml_type llama_ftype_get_default_type(llama_ftype ftype);20 21struct quantize_state_impl;22 23LLAMA_API quantize_state_impl * llama_quant_init(24 const llama_model * model,25 const llama_model_quantize_params * params);26 27LLAMA_API void llama_quant_free(quantize_state_impl * qs);28 29// Descriptor for constructing a mock model for quantization testing.30struct llama_quant_model_desc {31 const char * architecture;32 uint32_t n_embd;33 uint32_t n_ff;34 uint32_t n_layer;35 uint32_t n_head;36 uint32_t n_head_kv;37 uint32_t n_expert;38 uint32_t n_embd_head_k;39 uint32_t n_embd_head_v;40};41 42// Create a mock model from a metadata descriptor (for testing).43// The returned model must be freed with llama_model_free().44LLAMA_API llama_model * llama_quant_model_from_metadata(const llama_quant_model_desc * desc);45 46// Returns true if this tensor should be quantized (based on name, dims, params).47LLAMA_API bool llama_quant_tensor_allows_quantization(48 const quantize_state_impl * qs,49 const ggml_tensor * tensor);50 51// Compute quantization type assignments for a list of tensors.52// All tensors should be quantizable (use llama_quant_tensor_allows_quantization to filter).53// result_types: caller-allocated array of n_tensors elements, filled with assigned types.54LLAMA_API void llama_quant_compute_types(55 quantize_state_impl * qs,56 llama_ftype ftype,57 ggml_tensor ** tensors,58 ggml_type * result_types,59 size_t n_tensors);60 61//62// device memory querying63//64 65// "memory" as in physical memory for a buffer type, in bytes66struct llama_memory_breakdown_data {67 size_t model = 0; // memory allocated for the model68 size_t context = 0; // memory allocated for the context69 size_t compute = 0; // memory allocated for temporary compute buffers70 71 size_t total() const {72 return model + context + compute;73 }74};75 76struct llama_device_memory_data {77 int64_t total;78 int64_t free;79 llama_memory_breakdown_data mb;80};81 82// TODO: convert to C-style data structure83using llama_memory_breakdown = std::map<ggml_backend_buffer_type_t, llama_memory_breakdown_data>;84 85LLAMA_API int32_t llama_model_n_expert (const struct llama_model * model);86LLAMA_API int32_t llama_model_n_devices(const struct llama_model * model);87 88LLAMA_API ggml_backend_dev_t llama_model_get_device(const struct llama_model * model, int i);89 90LLAMA_API llama_memory_breakdown llama_get_memory_breakdown(const struct llama_context * ctx);91 