Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
server-models.h391 linesDownload Raw Back to server
1#pragma once2 3#include "common.h"4#include "download.h"5#include "preset.h"6#include "server-common.h"7#include "server-http.h"8#include "server-queue.h"9 10#include <mutex>11#include <condition_variable>12#include <functional>13#include <memory>14#include <optional>15#include <set>16#include <string>17#include <unordered_map>18 19/**20 * state diagram:21 *22 * DOWNLOADING ──► DOWNLOADED ──► (replaced by new instance)23 *24 * UNLOADED ──► LOADING ──► LOADED ◄──── SLEEPING25 *  ▲            │            │               ▲26 *  └───failed───┘            │               │27 *  ▲                         └──sleeping─────┘28 *  └────────unloaded─────────┘29 */30enum server_model_status {31    // TODO: also add downloading state when the logic is added32    SERVER_MODEL_STATUS_DOWNLOADING,33    SERVER_MODEL_STATUS_DOWNLOADED,34    SERVER_MODEL_STATUS_UNLOADED,35    SERVER_MODEL_STATUS_LOADING,36    SERVER_MODEL_STATUS_LOADED,37    SERVER_MODEL_STATUS_SLEEPING38};39 40enum server_model_source {41    SERVER_MODEL_SOURCE_PRESET,42    SERVER_MODEL_SOURCE_MODELS_DIR,43    SERVER_MODEL_SOURCE_CACHE,44};45 46enum server_child_mode {47    SERVER_CHILD_MODE_NORMAL,   // load the model and run normally48    SERVER_CHILD_MODE_DOWNLOAD, // download the model and exit49};50 51static std::string server_model_status_to_string(server_model_status status) {52    switch (status) {53        case SERVER_MODEL_STATUS_DOWNLOADING: return "downloading";54        case SERVER_MODEL_STATUS_DOWNLOADED:  return "downloaded";55        case SERVER_MODEL_STATUS_UNLOADED:    return "unloaded";56        case SERVER_MODEL_STATUS_LOADING:     return "loading";57        case SERVER_MODEL_STATUS_LOADED:      return "loaded";58        case SERVER_MODEL_STATUS_SLEEPING:    return "sleeping";59        default:                              return "unknown";60    }61}62 63static std::string server_model_source_to_string(server_model_source source) {64    switch (source) {65        case SERVER_MODEL_SOURCE_PRESET:     return "preset";66        case SERVER_MODEL_SOURCE_MODELS_DIR: return "models_dir";67        case SERVER_MODEL_SOURCE_CACHE:      return "cache";68        default:                             return "unknown";69    }70}71 72struct server_model_meta {73    server_model_source source = SERVER_MODEL_SOURCE_CACHE;74    common_preset preset;75    std::string name;76    std::set<std::string> aliases; // additional names that resolve to this model77    std::set<std::string> tags;    // informational tags, not used for routing78    int port = 0;79    server_model_status status = SERVER_MODEL_STATUS_UNLOADED;80    int64_t last_used = 0; // for LRU unloading81    std::vector<std::string> args; // args passed to the model instance, will be populated by render_args()82    json loaded_info; // info to be reflected via /v1/models endpoint ; if in DOWNLOADING state, it should contain download progress info83    json progress; // reflect load or download progress info, if any84    int exit_code = 0; // exit code of the model instance process (only valid if status == FAILED)85    int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown86    mtmd_caps multimodal; // multimodal capabilities87 88    bool is_ready() const {89        return status == SERVER_MODEL_STATUS_LOADED;90    }91 92    bool is_running() const {93        return status == SERVER_MODEL_STATUS_LOADED || status == SERVER_MODEL_STATUS_LOADING || status == SERVER_MODEL_STATUS_SLEEPING;94    }95 96    bool is_ready_or_sleep() const {97        return status == SERVER_MODEL_STATUS_LOADED || status == SERVER_MODEL_STATUS_SLEEPING;98    }99 100    bool is_failed() const {101        return status == SERVER_MODEL_STATUS_UNLOADED && exit_code != 0;102    }103 104    void update_args(common_preset_context & ctx_presets, std::string bin_path);105    void update_caps();106};107 108struct server_models_routes;109struct server_subproc;   // defined in server-models.cpp110struct server_lru_sched; // defined in server-models.cpp111 112struct server_models {113    friend struct server_models_routes;114    friend struct server_lru_sched;115 116private:117    struct instance_t {118        std::shared_ptr<server_subproc> subproc; // shared between main thread and monitoring thread119        std::thread th;120        server_model_meta meta;121        int req_count = 0; // number of active proxy requests122    };123 124    std::mutex mutex;125    std::condition_variable cv;126    std::map<std::string, instance_t> mapping;127 128    // for stopping models129    std::condition_variable cv_stop;130    std::set<std::string> stopping_models;131 132    // set to true while load_models() is executing a reload; load() will wait until clear133    bool is_reloading = false;134 135    // if true, the next get_meta() will trigger a reload of model list136    bool need_reload = false;137 138    // conv_id -> model name that currently serves its stream session, lets the resumable stream139    // routes go straight to the owning child instead of polling every one. populated when140    // proxy_request forwards a POST carrying an X-Conversation-Id. best effort: a stale entry just141    // makes the child answer not found and the client recovers. owns its lock, one mutex per struct142    struct conv_model_tracker {143        // returns the ticket of this registration, 0 when nothing was registered. erasing or144        // replacing the entry invalidates the ticket, which is how a stop cancels a request145        // parked in the model load wait146        uint64_t remember(const std::string & conv_id, const std::string & model) {147            if (conv_id.empty() || model.empty()) {148                return 0;149            }150            std::lock_guard<std::mutex> lock(mu);151            uint64_t ticket = next_ticket++;152            map[conv_id] = { model, ticket };153            return ticket;154        }155 156        // false means a stop erased the entry or a newer request replaced it157        bool alive(const std::string & conv_id, uint64_t ticket) {158            std::lock_guard<std::mutex> lock(mu);159            auto it = map.find(conv_id);160            return it != map.end() && it->second.ticket == ticket;161        }162 163        std::optional<std::string> lookup(const std::string & conv_id) {164            if (conv_id.empty()) {165                return std::nullopt;166            }167            std::lock_guard<std::mutex> lock(mu);168            auto it = map.find(conv_id);169            if (it == map.end()) {170                return std::nullopt;171            }172            return it->second.model;173        }174 175        void forget(const std::string & conv_id) {176            if (conv_id.empty()) {177                return;178            }179            std::lock_guard<std::mutex> lock(mu);180            map.erase(conv_id);181        }182 183      private:184        struct entry_t {185            std::string model;186            uint64_t    ticket;187        };188        std::mutex                               mu;189        uint64_t                                 next_ticket = 1;190        std::unordered_map<std::string, entry_t> map;191    };192 193    common_preset_context ctx_preset;194 195    common_params base_params;196    std::string bin_path;197    std::vector<std::string> base_env;198    common_preset base_preset; // base preset from llama-server CLI args199 200    // queue of requests waiting for a models_max slot201    std::unique_ptr<server_lru_sched> sched;202 203    // if true, add some delay to simulate works (useful for testing)204    bool debug_fake_timing = false;205 206    void update_meta(const std::string & name, const server_model_meta & meta);207 208    // unload least recently used models if the limit is reached209    void unload_lru();210 211    // not thread-safe, caller must hold mutex212    void add_model(server_model_meta && meta);213 214    // notify SSE clients215    void notify_sse(const std::string & event, const std::string & model_id, const json & data = nullptr);216 217public:218    // conv_id -> model tracker for the resumable stream routes, owns its lock219    conv_model_tracker conv_models;220 221    server_models(const common_params & params, int argc, char ** argv);222    ~server_models();223 224    server_response sse; // for real-time updates via SSE endpoint225 226    // (re-)load the list of models from various sources and prepare the metadata mapping227    // - if this is called the first time, simply populate the metadata228    // - if this is called subsequently (e.g. when refreshing from disk):229    //   - if a model is running but updated or removed from the source, it will be unloaded230    //   - if a model is not running, it will be added or updated according to the source231    void load_models();232 233    // check if a model instance exists (thread-safe)234    bool has_model(const std::string & name);235 236    // return a copy of model metadata (thread-safe)237    std::optional<server_model_meta> get_meta(const std::string & name);238 239    // return a copy of all model metadata (thread-safe)240    std::vector<server_model_meta> get_all_meta();241 242    struct load_options {243        server_child_mode mode = SERVER_CHILD_MODE_NORMAL;244        // used for spawning a downloading child process245        std::optional<server_model_meta> custom_meta = std::nullopt;246    };247 248    // load and unload model instances249    // these functions are thread-safe250    void load(const std::string & name);251    void load(const std::string & name, const load_options & opts);252    void unload(const std::string & name);253    void unload_all();254 255    struct update_status_args {256        server_model_status status;257        int exit_code = 0; // only valid if status == UNLOADED258        json loaded_info = nullptr;259        json progress = nullptr;260    };261    // update the status of a model instance (thread-safe)262    // also send SSE notification to /models/sse endpoint263    void update_status(const std::string & name, const update_status_args & args);264    void update_download_progress(const std::string & name, const common_download_progress & progress, bool done, bool ok = true);265 266    // remove a cache model from disk and update the list (thread-safe)267    // note: only cache models can be removed; returns false if the model doesn't exist or is not a cache model268    bool remove(const std::string & name);269 270    // wait until the model instance is fully loaded (thread-safe)271    // note: predicate is called while holding the lock272    // return when the model no longer in "loading" state273    void wait(const std::string & name, std::function<bool(const server_model_meta &)> predicate);274    void wait(std::unique_lock<std::mutex> & lk, const std::string & name, std::function<bool(const server_model_meta &)> predicate);275 276    // ensure the model is in ready state (thread-safe)277    // return false if model is ready278    // otherwise, load the model and blocking wait until it's ready, then return true (meta may need to be refreshed)279    // if models_max is reached, the request waits in a queue until a slot frees up280    // throws if the load fails, or if should_stop fires while waiting281    bool ensure_model_ready(const std::string & name, const std::function<bool()> & should_stop = nullptr);282 283    // proxy an HTTP request to the model instance284    server_http_res_ptr proxy_request(const server_http_req & req, const std::string & method, const std::string & name, bool update_last_used, bool detached = false);285 286    // handle message sent from server_child::notify_to_router()287    // raw input must starts with CMD_CHILD_TO_ROUTER_STATE, followed by a JSON string288    // this function is not thread-safe, must be called from instance's monitoring thread289    // payload per state:290    //     state = loading     -> payload = {} (TODO: add progress info)291    //     state = ready       -> payload = model_info (json), or {} if wakeup from sleeping292    //     state = sleeping    -> payload = {}293    void handle_child_state(const std::string & name, const std::string & raw_input);294};295 296struct server_child {297    // serializes the notify_to_router writes298    std::mutex mtx_stdout;299    std::atomic<bool> is_finished_downloading = false; // set by run_download300 301    // return true if the current process is a child server instance302    bool is_child();303    server_child_mode get_mode();304    int run_download(common_params & params);305 306    // register the shutdown_handler to be called by the router307    // return the monitoring thread (to be joined by the caller)308    std::thread setup(const std::function<void(int)> & shutdown_handler);309 310    // notify router server for status changes (e.g. loading, downloading, sleeping, etc.)311    // message will be handled by server_models::handle_child_state() on the router side312    void notify_to_router(const std::string & state_name, const json & payload);313};314 315struct server_models_routes {316    common_params params;317    json ui_settings = json::object();     // Primary: new name318    std::atomic<bool> stopping = false;    // for graceful disconnecting SSE clients during shutdown319    server_models models;320    server_models_routes(const common_params & params, int argc, char ** argv)321            : params(params), models(params, argc, argv) {322        const std::string & cfg = this->params.ui_config_json;323        if (!cfg.empty()) {324            try {325                json json_settings = json::parse(cfg);326                ui_settings = json_settings;327            } catch (const std::exception & e) {328                LOG_ERR("%s: failed to parse UI config: %s\n", __func__, e.what());329                throw;330            }331        }332        init_routes();333    }334 335    void init_routes();336    // handlers using lambda function, so that they can capture `this` without `std::bind`337    server_http_context::handler_t get_router_props;338    server_http_context::handler_t proxy_get;339    server_http_context::handler_t proxy_post;340    server_http_context::handler_t get_router_models;341    server_http_context::handler_t post_router_models_load;342    server_http_context::handler_t post_router_models_unload;343    // management API344    server_http_context::handler_t get_router_models_sse;345    server_http_context::handler_t post_router_models;346    server_http_context::handler_t del_router_models;347 348    // router side handlers for the resumable streaming routes. each resolves the child that owns349    // a conversation through the conv_id -> model map, no probing or fan out350    server_http_context::handler_t router_stream_get;351    server_http_context::handler_t router_streams_lookup;352    server_http_context::handler_t router_stream_delete;353};354 355/**356 * A simple HTTP proxy that forwards requests to another server357 * and relays the responses back.358 */359struct server_http_proxy : server_http_res {360    std::function<void()> cleanup = nullptr;361    server_http_proxy(const std::string & method,362                      const std::string & scheme,363                      const std::string & host,364                      int port,365                      const std::string & path,366                      const std::map<std::string, std::string> & headers,367                      const std::string & body,368                      const std::map<std::string, uploaded_file> & files,369                      const std::function<bool()> should_stop,370                      int32_t timeout_read,371                      int32_t timeout_write372                      );373    ~server_http_proxy() {374        if (cleanup_pipes) {375            cleanup_pipes();376        }377        if (cleanup) {378            cleanup();379        }380    }381private:382    std::function<void()> cleanup_pipes = nullptr;383    std::thread thread;384    struct msg_t {385        std::map<std::string, std::string> headers;386        int status = 0;387        std::string data;388        std::string content_type;389    };390};391 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai