Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
server-context.h183 linesDownload Raw Back to server
1#pragma once2 3#include "server-http.h"4#include "server-task.h"5#include "server-queue.h"6 7#include <nlohmann/json_fwd.hpp>8 9#include <cstddef>10#include <memory>11#include <set>12 13struct server_context_impl; // private implementation14 15struct server_context_meta {16    std::string build_info;17    std::string model_name;18    std::set<std::string> model_aliases;19    std::set<std::string> model_tags;20    std::string model_path;21    bool has_mtmd;22    bool has_inp_image;23    bool has_inp_audio;24    bool has_inp_video;25    json json_ui_settings;26    int slot_n_ctx;27    enum llama_pooling_type pooling_type;28 29    // chat params30    server_chat_params & chat_params;31    std::map<std::string, bool> chat_template_caps;32 33    // tokens34    std::string bos_token_str;35    std::string eos_token_str;36    llama_token fim_pre_token;37    llama_token fim_sub_token;38    llama_token fim_mid_token;39    llama_token fim_pad_token;40    llama_token fim_rep_token;41    llama_token fim_sep_token;42 43    // sampling44    std::vector<llama_logit_bias> logit_bias_eog;45 46    // model meta47    enum llama_vocab_type model_vocab_type;48    int32_t model_vocab_n_tokens;49    int32_t model_n_ctx_train;50    int32_t model_n_embd_inp;51    uint64_t model_n_params;52    uint64_t model_size;53    std::string model_ftype;54};55 56enum server_state {57    SERVER_STATE_DOWNLOADING,58    SERVER_STATE_LOADING,59    SERVER_STATE_READY,60    SERVER_STATE_SLEEPING,61};62 63static std::string server_state_to_str(server_state state) {64    switch (state) {65        case SERVER_STATE_DOWNLOADING: return "downloading";66        case SERVER_STATE_LOADING:     return "loading";67        case SERVER_STATE_READY:       return "ready";68        case SERVER_STATE_SLEEPING:    return "sleeping";69        default: GGML_ASSERT(false && "invalid server_state");70    }71}72 73static server_state server_state_from_str(const std::string & str) {74    if (str == "downloading") return SERVER_STATE_DOWNLOADING;75    if (str == "loading")     return SERVER_STATE_LOADING;76    if (str == "ready")       return SERVER_STATE_READY;77    if (str == "sleeping")    return SERVER_STATE_SLEEPING;78    GGML_ASSERT(false && "invalid server_state string");79}80 81using server_state_callback_t = std::function<void(server_state, json /* payload */)>;82 83struct server_context {84    std::unique_ptr<server_context_impl> impl;85 86    server_context();87    ~server_context();88 89    // load the model and initialize llama_context90    // returns true on success91    bool load_model(common_params & params);92 93    // this function will block main thread until termination94    void start_loop();95 96    // terminate main loop (will unblock start_loop)97    void terminate();98 99    // get the underlaying llama_context, can return nullptr if sleeping100    // not thread-safe, should only be used from the main thread101    llama_context * get_llama_context() const;102 103    // get a new response reader, used by CLI application104    server_response_reader get_response_reader();105 106    // get server metadata (read-only), can only be called after load_model()107    // not thread-safe, should only be used from the main thread108    server_context_meta get_meta() const;109 110    // note: must be set before load_model() is called111    void set_state_callback(server_state_callback_t callback);112};113 114 115// forward declarations116struct server_res_generator;117 118struct server_routes {119    server_routes(const common_params & params, server_context & ctx_server);120 121    void init_routes();122 123    // note: this is not thread-safe and can only when ctx_http.is_ready is false124    void update_meta(const server_context & ctx_server) {125        this->meta = std::make_unique<server_context_meta>(ctx_server.get_meta());126    }127 128    // handlers using lambda function, so that they can capture `this` without `std::bind`129    // they won't be called until ctx_http.is_ready is set to true130    server_http_context::handler_t get_health;131    server_http_context::handler_t get_metrics;132    server_http_context::handler_t get_slots;133    server_http_context::handler_t post_slots;134    server_http_context::handler_t get_props;135    server_http_context::handler_t post_props;136    server_http_context::handler_t post_infill;137    server_http_context::handler_t post_completions;138    server_http_context::handler_t post_completions_oai;139    server_http_context::handler_t post_chat_completions;140    server_http_context::handler_t post_chat_completions_tok;141    server_http_context::handler_t post_control;142    server_http_context::handler_t post_responses_oai;143    server_http_context::handler_t post_responses_tok_oai;144    server_http_context::handler_t post_transcriptions_oai;145    server_http_context::handler_t post_anthropic_messages;146    server_http_context::handler_t post_anthropic_count_tokens;147    server_http_context::handler_t post_apply_template;148    server_http_context::handler_t get_models;149    server_http_context::handler_t post_tokenize;150    server_http_context::handler_t post_detokenize;151    server_http_context::handler_t post_embeddings;152    server_http_context::handler_t post_embeddings_oai;153    server_http_context::handler_t post_rerank;154    server_http_context::handler_t get_lora_adapters;155    server_http_context::handler_t post_lora_adapters;156 157    // to be used in router mode158    json get_model_info() const;159 160private:161    std::unique_ptr<server_res_generator> handle_completions_impl(162            const server_http_req & req,163            server_task_type type,164            const json & data,165            const std::vector<raw_buffer> & files,166            task_response_type res_type);167    std::unique_ptr<server_res_generator> handle_slots_save(const server_http_req & req, int id_slot);168    std::unique_ptr<server_res_generator> handle_slots_restore(const server_http_req & req, int id_slot);169    std::unique_ptr<server_res_generator> handle_slots_erase(const server_http_req &, int id_slot);170    std::unique_ptr<server_res_generator> handle_embeddings_impl(const server_http_req & req, task_response_type res_type);171    std::unique_ptr<server_res_generator> handle_count_tokens(const llama_vocab * vocab, mtmd_context * mctx, const server_http_req & req, task_response_type res_type);172 173    // using unique_ptr to allow late initialization of const174    std::unique_ptr<const server_context_meta> meta;175 176    const common_params & params;177    const server_context_impl & ctx_server;178 179    server_queue & queue_tasks;180    server_response & queue_results;181    std::unique_ptr<server_res_generator> create_response(bool bypass_sleep = false);182};183 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai