Team Ai
Datasetpublic

echodict/llama.cpp

version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes479downloads
server-models.h223 linesDownload Raw Back to server
1#pragma once2 3#include "common.h"4#include "preset.h"5#include "server-common.h"6#include "server-http.h"7 8#include <mutex>9#include <condition_variable>10#include <functional>11#include <memory>12#include <set>13 14/**15 * state diagram:16 *17 * UNLOADED ──► LOADING ──► LOADED ◄──── SLEEPING18 *  ▲            │            │               ▲19 *  └───failed───┘            │               │20 *  ▲                         └──sleeping─────┘21 *  └────────unloaded─────────┘22 */23enum server_model_status {24    // TODO: also add downloading state when the logic is added25    SERVER_MODEL_STATUS_UNLOADED,26    SERVER_MODEL_STATUS_LOADING,27    SERVER_MODEL_STATUS_LOADED,28    SERVER_MODEL_STATUS_SLEEPING29};30 31static server_model_status server_model_status_from_string(const std::string & status_str) {32    if (status_str == "unloaded") {33        return SERVER_MODEL_STATUS_UNLOADED;34    }35    if (status_str == "loading") {36        return SERVER_MODEL_STATUS_LOADING;37    }38    if (status_str == "loaded") {39        return SERVER_MODEL_STATUS_LOADED;40    }41    if (status_str == "sleeping") {42        return SERVER_MODEL_STATUS_SLEEPING;43    }44    throw std::runtime_error("invalid server model status");45}46 47static std::string server_model_status_to_string(server_model_status status) {48    switch (status) {49        case SERVER_MODEL_STATUS_UNLOADED: return "unloaded";50        case SERVER_MODEL_STATUS_LOADING:  return "loading";51        case SERVER_MODEL_STATUS_LOADED:   return "loaded";52        case SERVER_MODEL_STATUS_SLEEPING: return "sleeping";53        default:                           return "unknown";54    }55}56 57struct server_model_meta {58    common_preset preset;59    std::string name;60    std::set<std::string> aliases; // additional names that resolve to this model61    std::set<std::string> tags;    // informational tags, not used for routing62    int port = 0;63    server_model_status status = SERVER_MODEL_STATUS_UNLOADED;64    int64_t last_used = 0; // for LRU unloading65    std::vector<std::string> args; // args passed to the model instance, will be populated by render_args()66    int exit_code = 0; // exit code of the model instance process (only valid if status == FAILED)67    int stop_timeout = 0; // seconds to wait before force-killing the model instance during shutdown68 69    bool is_ready() const {70        return status == SERVER_MODEL_STATUS_LOADED;71    }72 73    bool is_running() const {74        return status == SERVER_MODEL_STATUS_LOADED || status == SERVER_MODEL_STATUS_LOADING || status == SERVER_MODEL_STATUS_SLEEPING;75    }76 77    bool is_failed() const {78        return status == SERVER_MODEL_STATUS_UNLOADED && exit_code != 0;79    }80 81    void update_args(common_preset_context & ctx_presets, std::string bin_path);82};83 84struct subprocess_s;85 86struct server_models {87private:88    struct instance_t {89        std::shared_ptr<subprocess_s> subproc; // shared between main thread and monitoring thread90        std::thread th;91        server_model_meta meta;92        FILE * stdin_file = nullptr;93    };94 95    std::mutex mutex;96    std::condition_variable cv;97    std::map<std::string, instance_t> mapping;98 99    // for stopping models100    std::condition_variable cv_stop;101    std::set<std::string> stopping_models;102 103    common_preset_context ctx_preset;104 105    common_params base_params;106    std::string bin_path;107    std::vector<std::string> base_env;108    common_preset base_preset; // base preset from llama-server CLI args109 110    void update_meta(const std::string & name, const server_model_meta & meta);111 112    // unload least recently used models if the limit is reached113    void unload_lru();114 115    // not thread-safe, caller must hold mutex116    void add_model(server_model_meta && meta);117 118public:119    server_models(const common_params & params, int argc, char ** argv);120 121    void load_models();122 123    // check if a model instance exists (thread-safe)124    bool has_model(const std::string & name);125 126    // return a copy of model metadata (thread-safe)127    std::optional<server_model_meta> get_meta(const std::string & name);128 129    // return a copy of all model metadata (thread-safe)130    std::vector<server_model_meta> get_all_meta();131 132    // load and unload model instances133    // these functions are thread-safe134    void load(const std::string & name);135    void unload(const std::string & name);136    void unload_all();137 138    // update the status of a model instance (thread-safe)139    void update_status(const std::string & name, server_model_status status, int exit_code);140 141    // wait until the model instance is fully loaded (thread-safe)142    // return when the model no longer in "loading" state143    void wait_until_loading_finished(const std::string & name);144 145    // ensure the model is in ready state (thread-safe)146    // return false if model is ready147    // otherwise, load the model and blocking wait until it's ready, then return true (meta may need to be refreshed)148    bool ensure_model_ready(const std::string & name);149 150    // proxy an HTTP request to the model instance151    server_http_res_ptr proxy_request(const server_http_req & req, const std::string & method, const std::string & name, bool update_last_used);152 153    // return true if the current process is a child server instance154    static bool is_child_server();155 156    // notify the router server that a model instance is ready157    // return the monitoring thread (to be joined by the caller)158    static std::thread setup_child_server(const std::function<void(int)> & shutdown_handler);159 160    // notify the router server that the sleeping state has changed161    static void notify_router_sleeping_state(bool sleeping);162};163 164struct server_models_routes {165    common_params params;166    json webui_settings = json::object();167    server_models models;168    server_models_routes(const common_params & params, int argc, char ** argv)169            : params(params), models(params, argc, argv) {170        if (!this->params.webui_config_json.empty()) {171            try {172                webui_settings = json::parse(this->params.webui_config_json);173            } catch (const std::exception & e) {174                LOG_ERR("%s: failed to parse webui config: %s\n", __func__, e.what());175                throw;176            }177        }178        init_routes();179    }180 181    void init_routes();182    // handlers using lambda function, so that they can capture `this` without `std::bind`183    server_http_context::handler_t get_router_props;184    server_http_context::handler_t proxy_get;185    server_http_context::handler_t proxy_post;186    server_http_context::handler_t get_router_models;187    server_http_context::handler_t post_router_models_load;188    server_http_context::handler_t post_router_models_unload;189};190 191/**192 * A simple HTTP proxy that forwards requests to another server193 * and relays the responses back.194 */195struct server_http_proxy : server_http_res {196    std::function<void()> cleanup = nullptr;197public:198    server_http_proxy(const std::string & method,199                      const std::string & scheme,200                      const std::string & host,201                      int port,202                      const std::string & path,203                      const std::map<std::string, std::string> & headers,204                      const std::string & body,205                      const std::function<bool()> should_stop,206                      int32_t timeout_read,207                      int32_t timeout_write208                      );209    ~server_http_proxy() {210        if (cleanup) {211            cleanup();212        }213    }214private:215    std::thread thread;216    struct msg_t {217        std::map<std::string, std::string> headers;218        int status = 0;219        std::string data;220        std::string content_type;221    };222};223