echodict/llama.cpp
version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786
0479
1#pragma once2 3#include "common.h"4#include "preset.h"5#include "server-common.h"6#include "server-http.h"7 8#include <mutex>9#include <condition_variable>10#include <functional>11#include <memory>12#include <set>13 14/**15 * state diagram:16 *17 * UNLOADED ──► LOADING ──► LOADED ◄──── SLEEPING18 * ▲ │ │ ▲19 * └───failed───┘ │ │20 * ▲ └──sleeping─────┘21 * └────────unloaded─────────┘22 */23enum server_model_status {24 // TODO: also add downloading state when the logic is added25 SERVER_MODEL_STATUS_UNLOADED,26 SERVER_MODEL_STATUS_LOADING,27 SERVER_MODEL_STATUS_LOADED,28 SERVER_MODEL_STATUS_SLEEPING29};30 31static server_model_status server_model_status_from_string(const std::string & status_str) {32 if (status_str == "unloaded") {33 return SERVER_MODEL_STATUS_UNLOADED;34 }35 if (status_str == "loading") {36 return SERVER_MODEL_STATUS_LOADING;37 }38 if (status_str == "loaded") {39 return SERVER_MODEL_STATUS_LOADED;40 }41 if (status_str == "sleeping") {42 return SERVER_MODEL_STATUS_SLEEPING;43 }44 throw std::runtime_error("invalid server model status");45}46 47static std::string server_model_status_to_string(server_model_status status) {48 switch (status) {49 case SERVER_MODEL_STATUS_UNLOADED: return "unloaded";50 case SERVER_MODEL_STATUS_LOADING: return "loading";51 case SERVER_MODEL_STATUS_LOADED: return "loaded";52 case SERVER_MODEL_STATUS_SLEEPING: return "sleeping";53 default: return "unknown";54 }55}56 57struct server_model_meta {58 common_preset preset;59 std::string name;60 std::set<std::string> aliases; // additional names that resolve to this model61 std::set<std::string> tags; // informational tags, not used for routing62 int port = 0;63 server_model_status status = SERVER_MODEL_STATUS_UNLOADED;64 int64_t last_used = 0; // for LRU unloading65 std::vector<std::string> args; // args passed to the model instance, will be populated by render_args()66 int exit_code = 0; // exit code of the model instance process (only valid if status == FAILED)67 int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown68 69 bool is_ready() const {70 return status == SERVER_MODEL_STATUS_LOADED;71 }72 73 bool is_running() const {74 return status == SERVER_MODEL_STATUS_LOADED || status == SERVER_MODEL_STATUS_LOADING || status == SERVER_MODEL_STATUS_SLEEPING;75 }76 77 bool is_failed() const {78 return status == SERVER_MODEL_STATUS_UNLOADED && exit_code != 0;79 }80 81 void update_args(common_preset_context & ctx_presets, std::string bin_path);82};83 84struct subprocess_s;85 86struct server_models {87private:88 struct instance_t {89 std::shared_ptr<subprocess_s> subproc; // shared between main thread and monitoring thread90 std::thread th;91 server_model_meta meta;92 FILE * stdin_file = nullptr;93 };94 95 std::mutex mutex;96 std::condition_variable cv;97 std::map<std::string, instance_t> mapping;98 99 // for stopping models100 std::condition_variable cv_stop;101 std::set<std::string> stopping_models;102 103 common_preset_context ctx_preset;104 105 common_params base_params;106 std::string bin_path;107 std::vector<std::string> base_env;108 common_preset base_preset; // base preset from llama-server CLI args109 110 void update_meta(const std::string & name, const server_model_meta & meta);111 112 // unload least recently used models if the limit is reached113 void unload_lru();114 115 // not thread-safe, caller must hold mutex116 void add_model(server_model_meta && meta);117 118public:119 server_models(const common_params & params, int argc, char ** argv);120 121 void load_models();122 123 // check if a model instance exists (thread-safe)124 bool has_model(const std::string & name);125 126 // return a copy of model metadata (thread-safe)127 std::optional<server_model_meta> get_meta(const std::string & name);128 129 // return a copy of all model metadata (thread-safe)130 std::vector<server_model_meta> get_all_meta();131 132 // load and unload model instances133 // these functions are thread-safe134 void load(const std::string & name);135 void unload(const std::string & name);136 void unload_all();137 138 // update the status of a model instance (thread-safe)139 void update_status(const std::string & name, server_model_status status, int exit_code);140 141 // wait until the model instance is fully loaded (thread-safe)142 // return when the model no longer in "loading" state143 void wait_until_loading_finished(const std::string & name);144 145 // ensure the model is in ready state (thread-safe)146 // return false if model is ready147 // otherwise, load the model and blocking wait until it's ready, then return true (meta may need to be refreshed)148 bool ensure_model_ready(const std::string & name);149 150 // proxy an HTTP request to the model instance151 server_http_res_ptr proxy_request(const server_http_req & req, const std::string & method, const std::string & name, bool update_last_used);152 153 // return true if the current process is a child server instance154 static bool is_child_server();155 156 // notify the router server that a model instance is ready157 // return the monitoring thread (to be joined by the caller)158 static std::thread setup_child_server(const std::function<void(int)> & shutdown_handler);159 160 // notify the router server that the sleeping state has changed161 static void notify_router_sleeping_state(bool sleeping);162};163 164struct server_models_routes {165 common_params params;166 json webui_settings = json::object();167 server_models models;168 server_models_routes(const common_params & params, int argc, char ** argv)169 : params(params), models(params, argc, argv) {170 if (!this->params.webui_config_json.empty()) {171 try {172 webui_settings = json::parse(this->params.webui_config_json);173 } catch (const std::exception & e) {174 LOG_ERR("%s: failed to parse webui config: %s\n", __func__, e.what());175 throw;176 }177 }178 init_routes();179 }180 181 void init_routes();182 // handlers using lambda function, so that they can capture `this` without `std::bind`183 server_http_context::handler_t get_router_props;184 server_http_context::handler_t proxy_get;185 server_http_context::handler_t proxy_post;186 server_http_context::handler_t get_router_models;187 server_http_context::handler_t post_router_models_load;188 server_http_context::handler_t post_router_models_unload;189};190 191/**192 * A simple HTTP proxy that forwards requests to another server193 * and relays the responses back.194 */195struct server_http_proxy : server_http_res {196 std::function<void()> cleanup = nullptr;197public:198 server_http_proxy(const std::string & method,199 const std::string & scheme,200 const std::string & host,201 int port,202 const std::string & path,203 const std::map<std::string, std::string> & headers,204 const std::string & body,205 const std::function<bool()> should_stop,206 int32_t timeout_read,207 int32_t timeout_write208 );209 ~server_http_proxy() {210 if (cleanup) {211 cleanup();212 }213 }214private:215 std::thread thread;216 struct msg_t {217 std::map<std::string, std::string> headers;218 int status = 0;219 std::string data;220 std::string content_type;221 };222};223 