Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
llama-cparams.h38 linesDownload Raw Back to src
1#pragma once2 3#include "llama.h"4 5#include <cstdint>6 7struct llama_cparams {8    uint32_t n_ctx;           // context size used during inference9    uint32_t n_batch;10    uint32_t n_ubatch;11    uint32_t n_seq_max;12    int      n_threads;       // number of threads to use for generation13    int      n_threads_batch; // number of threads to use for batch processing14 15    float rope_freq_base;16    float rope_freq_scale;17 18    uint32_t n_ctx_orig_yarn;19    // These hyperparameters are not exposed in GGUF, because all20    // existing YaRN models use the same values for them.21    float yarn_ext_factor;22    float yarn_attn_factor;23    float yarn_beta_fast;24    float yarn_beta_slow;25    float defrag_thold;26 27    bool embeddings;28    bool causal_attn;29    bool offload_kqv;30    bool flash_attn;31    bool no_perf;32 33    enum llama_pooling_type pooling_type;34 35    ggml_backend_sched_eval_callback cb_eval;36    void * cb_eval_user_data;37};38