KBaba7/llama.cpp
0
1#pragma once2 3#include "llama.h"4 5#include <cstdint>6 7struct llama_cparams {8 uint32_t n_ctx; // context size used during inference9 uint32_t n_batch;10 uint32_t n_ubatch;11 uint32_t n_seq_max;12 int n_threads; // number of threads to use for generation13 int n_threads_batch; // number of threads to use for batch processing14 15 float rope_freq_base;16 float rope_freq_scale;17 18 uint32_t n_ctx_orig_yarn;19 // These hyperparameters are not exposed in GGUF, because all20 // existing YaRN models use the same values for them.21 float yarn_ext_factor;22 float yarn_attn_factor;23 float yarn_beta_fast;24 float yarn_beta_slow;25 float defrag_thold;26 27 bool embeddings;28 bool causal_attn;29 bool offload_kqv;30 bool flash_attn;31 bool no_perf;32 33 enum llama_pooling_type pooling_type;34 35 ggml_backend_sched_eval_callback cb_eval;36 void * cb_eval_user_data;37};38 