Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#pragma once2 3#include "common.h"4#include "download.h"5#include "preset.h"6#include "server-common.h"7#include "server-http.h"8#include "server-queue.h"9 10#include <mutex>11#include <condition_variable>12#include <functional>13#include <memory>14#include <optional>15#include <set>16#include <string>17#include <unordered_map>18 19/**20 * state diagram:21 *22 * DOWNLOADING ──► DOWNLOADED ──► (replaced by new instance)23 *24 * UNLOADED ──► LOADING ──► LOADED ◄──── SLEEPING25 * ▲ │ │ ▲26 * └───failed───┘ │ │27 * ▲ └──sleeping─────┘28 * └────────unloaded─────────┘29 */30enum server_model_status {31 // TODO: also add downloading state when the logic is added32 SERVER_MODEL_STATUS_DOWNLOADING,33 SERVER_MODEL_STATUS_DOWNLOADED,34 SERVER_MODEL_STATUS_UNLOADED,35 SERVER_MODEL_STATUS_LOADING,36 SERVER_MODEL_STATUS_LOADED,37 SERVER_MODEL_STATUS_SLEEPING38};39 40enum server_model_source {41 SERVER_MODEL_SOURCE_PRESET,42 SERVER_MODEL_SOURCE_MODELS_DIR,43 SERVER_MODEL_SOURCE_CACHE,44};45 46enum server_child_mode {47 SERVER_CHILD_MODE_NORMAL, // load the model and run normally48 SERVER_CHILD_MODE_DOWNLOAD, // download the model and exit49};50 51static std::string server_model_status_to_string(server_model_status status) {52 switch (status) {53 case SERVER_MODEL_STATUS_DOWNLOADING: return "downloading";54 case SERVER_MODEL_STATUS_DOWNLOADED: return "downloaded";55 case SERVER_MODEL_STATUS_UNLOADED: return "unloaded";56 case SERVER_MODEL_STATUS_LOADING: return "loading";57 case SERVER_MODEL_STATUS_LOADED: return "loaded";58 case SERVER_MODEL_STATUS_SLEEPING: return "sleeping";59 default: return "unknown";60 }61}62 63static std::string server_model_source_to_string(server_model_source source) {64 switch (source) {65 case SERVER_MODEL_SOURCE_PRESET: return "preset";66 case SERVER_MODEL_SOURCE_MODELS_DIR: return "models_dir";67 case SERVER_MODEL_SOURCE_CACHE: return "cache";68 default: return "unknown";69 }70}71 72struct server_model_meta {73 server_model_source source = SERVER_MODEL_SOURCE_CACHE;74 common_preset preset;75 std::string name;76 std::set<std::string> aliases; // additional names that resolve to this model77 std::set<std::string> tags; // informational tags, not used for routing78 int port = 0;79 server_model_status status = SERVER_MODEL_STATUS_UNLOADED;80 int64_t last_used = 0; // for LRU unloading81 std::vector<std::string> args; // args passed to the model instance, will be populated by render_args()82 json loaded_info; // info to be reflected via /v1/models endpoint ; if in DOWNLOADING state, it should contain download progress info83 json progress; // reflect load or download progress info, if any84 int exit_code = 0; // exit code of the model instance process (only valid if status == FAILED)85 int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown86 mtmd_caps multimodal; // multimodal capabilities87 88 bool is_ready() const {89 return status == SERVER_MODEL_STATUS_LOADED;90 }91 92 bool is_running() const {93 return status == SERVER_MODEL_STATUS_LOADED || status == SERVER_MODEL_STATUS_LOADING || status == SERVER_MODEL_STATUS_SLEEPING;94 }95 96 bool is_ready_or_sleep() const {97 return status == SERVER_MODEL_STATUS_LOADED || status == SERVER_MODEL_STATUS_SLEEPING;98 }99 100 bool is_failed() const {101 return status == SERVER_MODEL_STATUS_UNLOADED && exit_code != 0;102 }103 104 void update_args(common_preset_context & ctx_presets, std::string bin_path);105 void update_caps();106};107 108struct server_models_routes;109struct server_subproc; // defined in server-models.cpp110struct server_lru_sched; // defined in server-models.cpp111 112struct server_models {113 friend struct server_models_routes;114 friend struct server_lru_sched;115 116private:117 struct instance_t {118 std::shared_ptr<server_subproc> subproc; // shared between main thread and monitoring thread119 std::thread th;120 server_model_meta meta;121 int req_count = 0; // number of active proxy requests122 };123 124 std::mutex mutex;125 std::condition_variable cv;126 std::map<std::string, instance_t> mapping;127 128 // for stopping models129 std::condition_variable cv_stop;130 std::set<std::string> stopping_models;131 132 // set to true while load_models() is executing a reload; load() will wait until clear133 bool is_reloading = false;134 135 // if true, the next get_meta() will trigger a reload of model list136 bool need_reload = false;137 138 // conv_id -> model name that currently serves its stream session, lets the resumable stream139 // routes go straight to the owning child instead of polling every one. populated when140 // proxy_request forwards a POST carrying an X-Conversation-Id. best effort: a stale entry just141 // makes the child answer not found and the client recovers. owns its lock, one mutex per struct142 struct conv_model_tracker {143 // returns the ticket of this registration, 0 when nothing was registered. erasing or144 // replacing the entry invalidates the ticket, which is how a stop cancels a request145 // parked in the model load wait146 uint64_t remember(const std::string & conv_id, const std::string & model) {147 if (conv_id.empty() || model.empty()) {148 return 0;149 }150 std::lock_guard<std::mutex> lock(mu);151 uint64_t ticket = next_ticket++;152 map[conv_id] = { model, ticket };153 return ticket;154 }155 156 // false means a stop erased the entry or a newer request replaced it157 bool alive(const std::string & conv_id, uint64_t ticket) {158 std::lock_guard<std::mutex> lock(mu);159 auto it = map.find(conv_id);160 return it != map.end() && it->second.ticket == ticket;161 }162 163 std::optional<std::string> lookup(const std::string & conv_id) {164 if (conv_id.empty()) {165 return std::nullopt;166 }167 std::lock_guard<std::mutex> lock(mu);168 auto it = map.find(conv_id);169 if (it == map.end()) {170 return std::nullopt;171 }172 return it->second.model;173 }174 175 void forget(const std::string & conv_id) {176 if (conv_id.empty()) {177 return;178 }179 std::lock_guard<std::mutex> lock(mu);180 map.erase(conv_id);181 }182 183 private:184 struct entry_t {185 std::string model;186 uint64_t ticket;187 };188 std::mutex mu;189 uint64_t next_ticket = 1;190 std::unordered_map<std::string, entry_t> map;191 };192 193 common_preset_context ctx_preset;194 195 common_params base_params;196 std::string bin_path;197 std::vector<std::string> base_env;198 common_preset base_preset; // base preset from llama-server CLI args199 200 // queue of requests waiting for a models_max slot201 std::unique_ptr<server_lru_sched> sched;202 203 // if true, add some delay to simulate works (useful for testing)204 bool debug_fake_timing = false;205 206 void update_meta(const std::string & name, const server_model_meta & meta);207 208 // unload least recently used models if the limit is reached209 void unload_lru();210 211 // not thread-safe, caller must hold mutex212 void add_model(server_model_meta && meta);213 214 // notify SSE clients215 void notify_sse(const std::string & event, const std::string & model_id, const json & data = nullptr);216 217public:218 // conv_id -> model tracker for the resumable stream routes, owns its lock219 conv_model_tracker conv_models;220 221 server_models(const common_params & params, int argc, char ** argv);222 ~server_models();223 224 server_response sse; // for real-time updates via SSE endpoint225 226 // (re-)load the list of models from various sources and prepare the metadata mapping227 // - if this is called the first time, simply populate the metadata228 // - if this is called subsequently (e.g. when refreshing from disk):229 // - if a model is running but updated or removed from the source, it will be unloaded230 // - if a model is not running, it will be added or updated according to the source231 void load_models();232 233 // check if a model instance exists (thread-safe)234 bool has_model(const std::string & name);235 236 // return a copy of model metadata (thread-safe)237 std::optional<server_model_meta> get_meta(const std::string & name);238 239 // return a copy of all model metadata (thread-safe)240 std::vector<server_model_meta> get_all_meta();241 242 struct load_options {243 server_child_mode mode = SERVER_CHILD_MODE_NORMAL;244 // used for spawning a downloading child process245 std::optional<server_model_meta> custom_meta = std::nullopt;246 };247 248 // load and unload model instances249 // these functions are thread-safe250 void load(const std::string & name);251 void load(const std::string & name, const load_options & opts);252 void unload(const std::string & name);253 void unload_all();254 255 struct update_status_args {256 server_model_status status;257 int exit_code = 0; // only valid if status == UNLOADED258 json loaded_info = nullptr;259 json progress = nullptr;260 };261 // update the status of a model instance (thread-safe)262 // also send SSE notification to /models/sse endpoint263 void update_status(const std::string & name, const update_status_args & args);264 void update_download_progress(const std::string & name, const common_download_progress & progress, bool done, bool ok = true);265 266 // remove a cache model from disk and update the list (thread-safe)267 // note: only cache models can be removed; returns false if the model doesn't exist or is not a cache model268 bool remove(const std::string & name);269 270 // wait until the model instance is fully loaded (thread-safe)271 // note: predicate is called while holding the lock272 // return when the model no longer in "loading" state273 void wait(const std::string & name, std::function<bool(const server_model_meta &)> predicate);274 void wait(std::unique_lock<std::mutex> & lk, const std::string & name, std::function<bool(const server_model_meta &)> predicate);275 276 // ensure the model is in ready state (thread-safe)277 // return false if model is ready278 // otherwise, load the model and blocking wait until it's ready, then return true (meta may need to be refreshed)279 // if models_max is reached, the request waits in a queue until a slot frees up280 // throws if the load fails, or if should_stop fires while waiting281 bool ensure_model_ready(const std::string & name, const std::function<bool()> & should_stop = nullptr);282 283 // proxy an HTTP request to the model instance284 server_http_res_ptr proxy_request(const server_http_req & req, const std::string & method, const std::string & name, bool update_last_used, bool detached = false);285 286 // handle message sent from server_child::notify_to_router()287 // raw input must starts with CMD_CHILD_TO_ROUTER_STATE, followed by a JSON string288 // this function is not thread-safe, must be called from instance's monitoring thread289 // payload per state:290 // state = loading -> payload = {} (TODO: add progress info)291 // state = ready -> payload = model_info (json), or {} if wakeup from sleeping292 // state = sleeping -> payload = {}293 void handle_child_state(const std::string & name, const std::string & raw_input);294};295 296struct server_child {297 // serializes the notify_to_router writes298 std::mutex mtx_stdout;299 std::atomic<bool> is_finished_downloading = false; // set by run_download300 301 // return true if the current process is a child server instance302 bool is_child();303 server_child_mode get_mode();304 int run_download(common_params & params);305 306 // register the shutdown_handler to be called by the router307 // return the monitoring thread (to be joined by the caller)308 std::thread setup(const std::function<void(int)> & shutdown_handler);309 310 // notify router server for status changes (e.g. loading, downloading, sleeping, etc.)311 // message will be handled by server_models::handle_child_state() on the router side312 void notify_to_router(const std::string & state_name, const json & payload);313};314 315struct server_models_routes {316 common_params params;317 json ui_settings = json::object(); // Primary: new name318 std::atomic<bool> stopping = false; // for graceful disconnecting SSE clients during shutdown319 server_models models;320 server_models_routes(const common_params & params, int argc, char ** argv)321 : params(params), models(params, argc, argv) {322 const std::string & cfg = this->params.ui_config_json;323 if (!cfg.empty()) {324 try {325 json json_settings = json::parse(cfg);326 ui_settings = json_settings;327 } catch (const std::exception & e) {328 LOG_ERR("%s: failed to parse UI config: %s\n", __func__, e.what());329 throw;330 }331 }332 init_routes();333 }334 335 void init_routes();336 // handlers using lambda function, so that they can capture `this` without `std::bind`337 server_http_context::handler_t get_router_props;338 server_http_context::handler_t proxy_get;339 server_http_context::handler_t proxy_post;340 server_http_context::handler_t get_router_models;341 server_http_context::handler_t post_router_models_load;342 server_http_context::handler_t post_router_models_unload;343 // management API344 server_http_context::handler_t get_router_models_sse;345 server_http_context::handler_t post_router_models;346 server_http_context::handler_t del_router_models;347 348 // router side handlers for the resumable streaming routes. each resolves the child that owns349 // a conversation through the conv_id -> model map, no probing or fan out350 server_http_context::handler_t router_stream_get;351 server_http_context::handler_t router_streams_lookup;352 server_http_context::handler_t router_stream_delete;353};354 355/**356 * A simple HTTP proxy that forwards requests to another server357 * and relays the responses back.358 */359struct server_http_proxy : server_http_res {360 std::function<void()> cleanup = nullptr;361 server_http_proxy(const std::string & method,362 const std::string & scheme,363 const std::string & host,364 int port,365 const std::string & path,366 const std::map<std::string, std::string> & headers,367 const std::string & body,368 const std::map<std::string, uploaded_file> & files,369 const std::function<bool()> should_stop,370 int32_t timeout_read,371 int32_t timeout_write372 );373 ~server_http_proxy() {374 if (cleanup_pipes) {375 cleanup_pipes();376 }377 if (cleanup) {378 cleanup();379 }380 }381private:382 std::function<void()> cleanup_pipes = nullptr;383 std::thread thread;384 struct msg_t {385 std::map<std::string, std::string> headers;386 int status = 0;387 std::string data;388 std::string content_type;389 };390};391 