Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#pragma once2 3#include "server-http.h"4#include "server-task.h"5#include "server-queue.h"6 7#include <nlohmann/json_fwd.hpp>8 9#include <cstddef>10#include <memory>11#include <set>12 13struct server_context_impl; // private implementation14 15struct server_context_meta {16 std::string build_info;17 std::string model_name;18 std::set<std::string> model_aliases;19 std::set<std::string> model_tags;20 std::string model_path;21 bool has_mtmd;22 bool has_inp_image;23 bool has_inp_audio;24 bool has_inp_video;25 json json_ui_settings;26 int slot_n_ctx;27 enum llama_pooling_type pooling_type;28 29 // chat params30 server_chat_params & chat_params;31 std::map<std::string, bool> chat_template_caps;32 33 // tokens34 std::string bos_token_str;35 std::string eos_token_str;36 llama_token fim_pre_token;37 llama_token fim_sub_token;38 llama_token fim_mid_token;39 llama_token fim_pad_token;40 llama_token fim_rep_token;41 llama_token fim_sep_token;42 43 // sampling44 std::vector<llama_logit_bias> logit_bias_eog;45 46 // model meta47 enum llama_vocab_type model_vocab_type;48 int32_t model_vocab_n_tokens;49 int32_t model_n_ctx_train;50 int32_t model_n_embd_inp;51 uint64_t model_n_params;52 uint64_t model_size;53 std::string model_ftype;54};55 56enum server_state {57 SERVER_STATE_DOWNLOADING,58 SERVER_STATE_LOADING,59 SERVER_STATE_READY,60 SERVER_STATE_SLEEPING,61};62 63static std::string server_state_to_str(server_state state) {64 switch (state) {65 case SERVER_STATE_DOWNLOADING: return "downloading";66 case SERVER_STATE_LOADING: return "loading";67 case SERVER_STATE_READY: return "ready";68 case SERVER_STATE_SLEEPING: return "sleeping";69 default: GGML_ASSERT(false && "invalid server_state");70 }71}72 73static server_state server_state_from_str(const std::string & str) {74 if (str == "downloading") return SERVER_STATE_DOWNLOADING;75 if (str == "loading") return SERVER_STATE_LOADING;76 if (str == "ready") return SERVER_STATE_READY;77 if (str == "sleeping") return SERVER_STATE_SLEEPING;78 GGML_ASSERT(false && "invalid server_state string");79}80 81using server_state_callback_t = std::function<void(server_state, json /* payload */)>;82 83struct server_context {84 std::unique_ptr<server_context_impl> impl;85 86 server_context();87 ~server_context();88 89 // load the model and initialize llama_context90 // returns true on success91 bool load_model(common_params & params);92 93 // this function will block main thread until termination94 void start_loop();95 96 // terminate main loop (will unblock start_loop)97 void terminate();98 99 // get the underlaying llama_context, can return nullptr if sleeping100 // not thread-safe, should only be used from the main thread101 llama_context * get_llama_context() const;102 103 // get a new response reader, used by CLI application104 server_response_reader get_response_reader();105 106 // get server metadata (read-only), can only be called after load_model()107 // not thread-safe, should only be used from the main thread108 server_context_meta get_meta() const;109 110 // note: must be set before load_model() is called111 void set_state_callback(server_state_callback_t callback);112};113 114 115// forward declarations116struct server_res_generator;117 118struct server_routes {119 server_routes(const common_params & params, server_context & ctx_server);120 121 void init_routes();122 123 // note: this is not thread-safe and can only when ctx_http.is_ready is false124 void update_meta(const server_context & ctx_server) {125 this->meta = std::make_unique<server_context_meta>(ctx_server.get_meta());126 }127 128 // handlers using lambda function, so that they can capture `this` without `std::bind`129 // they won't be called until ctx_http.is_ready is set to true130 server_http_context::handler_t get_health;131 server_http_context::handler_t get_metrics;132 server_http_context::handler_t get_slots;133 server_http_context::handler_t post_slots;134 server_http_context::handler_t get_props;135 server_http_context::handler_t post_props;136 server_http_context::handler_t post_infill;137 server_http_context::handler_t post_completions;138 server_http_context::handler_t post_completions_oai;139 server_http_context::handler_t post_chat_completions;140 server_http_context::handler_t post_chat_completions_tok;141 server_http_context::handler_t post_control;142 server_http_context::handler_t post_responses_oai;143 server_http_context::handler_t post_responses_tok_oai;144 server_http_context::handler_t post_transcriptions_oai;145 server_http_context::handler_t post_anthropic_messages;146 server_http_context::handler_t post_anthropic_count_tokens;147 server_http_context::handler_t post_apply_template;148 server_http_context::handler_t get_models;149 server_http_context::handler_t post_tokenize;150 server_http_context::handler_t post_detokenize;151 server_http_context::handler_t post_embeddings;152 server_http_context::handler_t post_embeddings_oai;153 server_http_context::handler_t post_rerank;154 server_http_context::handler_t get_lora_adapters;155 server_http_context::handler_t post_lora_adapters;156 157 // to be used in router mode158 json get_model_info() const;159 160private:161 std::unique_ptr<server_res_generator> handle_completions_impl(162 const server_http_req & req,163 server_task_type type,164 const json & data,165 const std::vector<raw_buffer> & files,166 task_response_type res_type);167 std::unique_ptr<server_res_generator> handle_slots_save(const server_http_req & req, int id_slot);168 std::unique_ptr<server_res_generator> handle_slots_restore(const server_http_req & req, int id_slot);169 std::unique_ptr<server_res_generator> handle_slots_erase(const server_http_req &, int id_slot);170 std::unique_ptr<server_res_generator> handle_embeddings_impl(const server_http_req & req, task_response_type res_type);171 std::unique_ptr<server_res_generator> handle_count_tokens(const llama_vocab * vocab, mtmd_context * mctx, const server_http_req & req, task_response_type res_type);172 173 // using unique_ptr to allow late initialization of const174 std::unique_ptr<const server_context_meta> meta;175 176 const common_params & params;177 const server_context_impl & ctx_server;178 179 server_queue & queue_tasks;180 server_response & queue_results;181 std::unique_ptr<server_res_generator> create_response(bool bypass_sleep = false);182};183 