echodict/llama.cpp
version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786
0479
1#pragma once2 3#include "llama.h"4 5#include <cstdint>6 7#define LLAMA_MAX_SEQ 2568 9struct llama_cparams {10 uint32_t n_ctx; // context size used during inference11 uint32_t n_ctx_seq; // context for a single sequence12 uint32_t n_batch;13 uint32_t n_ubatch;14 uint32_t n_seq_max;15 int32_t n_threads; // number of threads to use for generation16 int32_t n_threads_batch; // number of threads to use for batch processing17 18 float rope_freq_base;19 float rope_freq_scale;20 21 uint32_t n_ctx_orig_yarn;22 // These hyperparameters are not exposed in GGUF, because all23 // existing YaRN models use the same values for them.24 float yarn_ext_factor;25 float yarn_attn_factor;26 float yarn_beta_fast;27 float yarn_beta_slow;28 29 bool embeddings;30 bool causal_attn;31 bool offload_kqv;32 bool flash_attn;33 bool auto_fa;34 bool fused_gdn_ar; // use fused gated delta net (autoregressive)35 bool fused_gdn_ch; // use fused gated delta net (chunked)36 bool auto_fgdn;37 bool no_perf;38 bool warmup;39 bool op_offload;40 bool kv_unified;41 bool pipeline_parallel;42 43 enum llama_pooling_type pooling_type;44 45 ggml_backend_sched_eval_callback cb_eval;46 void * cb_eval_user_data;47};48 