Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
speculative.h29 linesDownload Raw Back to common
1#pragma once2 3#include "llama.h"4#include "common.h"5 6struct common_speculative;7 8struct common_speculative_params {9    int n_draft = 16;  // max drafted tokens10    int n_reuse = 256;11 12    float p_min = 0.9f; // min probabiliy required to accept a token in the draft13};14 15struct common_speculative * common_speculative_init(struct llama_context * ctx_dft);16 17void common_speculative_free(struct common_speculative * spec);18 19bool common_speculative_are_compatible(20        const struct llama_context * ctx_tgt,21        const struct llama_context * ctx_dft);22 23// sample up to n_draft tokens and add them to the batch using the draft model24llama_tokens common_speculative_gen_draft(25               struct common_speculative * spec,26        struct common_speculative_params   params,27                      const llama_tokens & prompt,28                             llama_token   id_last);29