Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#include "server-context.h"2#include "server-http.h"3#include "server-models.h"4#include "server-cors-proxy.h"5#include "server-stream.h"6#include "server-tools.h"7 8#include "arg.h"9#include "build-info.h"10#include "common.h"11#include "fit.h"12#include "llama.h"13#include "log.h"14 15#include <atomic>16#include <clocale>17#include <exception>18#include <signal.h>19#include <thread> // for std::thread::hardware_concurrency20 21#if defined(_WIN32)22#include <windows.h>23#endif24 25static std::function<void(int)> shutdown_handler;26static std::atomic_flag is_terminating = ATOMIC_FLAG_INIT;27 28static inline void signal_handler(int signal) {29 if (is_terminating.test_and_set()) {30 // in case it hangs, we can force terminate the server by hitting Ctrl+C twice31 // this is for better developer experience, we can remove when the server is stable enough32 fprintf(stderr, "Received second interrupt, terminating immediately.\n");33 exit(1);34 }35 36 shutdown_handler(signal);37}38 39// satisfies -Wmissing-declarations (used by llama command)40int llama_server(int argc, char ** argv);41 42// to be used via CLI (argc / argv are used by router mode only)43int llama_server(common_params & params, int argc, char ** argv);44void llama_server_terminate();45void llama_server_terminate() {46 if (shutdown_handler) {47 shutdown_handler(0);48 }49}50 51 52// wrapper function that handles exceptions and logs errors53// this is to make sure handler_t never throws exceptions; instead, it returns an error response54static server_http_context::handler_t ex_wrapper(server_http_context::handler_t func) {55 return [func = std::move(func)](const server_http_req & req) -> server_http_res_ptr {56 std::string message;57 error_type error;58 try {59 return func(req);60 } catch (const std::invalid_argument & e) {61 // treat invalid_argument as invalid request (400)62 error = ERROR_TYPE_INVALID_REQUEST;63 message = e.what();64 } catch (const std::exception & e) {65 // treat other exceptions as server error (500)66 error = ERROR_TYPE_SERVER;67 message = e.what();68 } catch (...) {69 error = ERROR_TYPE_SERVER;70 message = "unknown error";71 }72 73 auto res = std::make_unique<server_http_res>();74 res->status = 500;75 try {76 json error_data = format_error_response(message, error);77 res->status = json_value(error_data, "code", 500);78 res->data = safe_json_to_str({{ "error", error_data }});79 SRV_WRN("got exception: %s\n", res->data.c_str());80 } catch (const std::exception & e) {81 SRV_ERR("got another exception: %s | while handling exception: %s\n", e.what(), message.c_str());82 res->data = "Internal Server Error";83 }84 return res;85 };86}87 88int llama_server(int argc, char ** argv) {89 std::setlocale(LC_NUMERIC, "C");90 91#ifndef _WIN3292 // Ignore SIGPIPE so the server does not crash if a child (MCP server, tools runtime) exits while we are writing to its stdin93 signal(SIGPIPE, SIG_IGN);94#endif95 96 // own arguments required by this example97 common_params params;98 99 common_init();100 101 // start the stream session manager GC right after common init, before any HTTP route can102 // touch it. lifecycle is symmetric, stop_gc() runs in clean_up() before backend free103 server_stream_session_manager_start();104 105 if (!common_params_parse(argc, argv, params, LLAMA_EXAMPLE_SERVER)) {106 return 1;107 }108 109 llama_backend_init();110 llama_numa_init(params.numa);111 112 return llama_server(params, argc, argv);113}114 115int llama_server(common_params & params, int argc, char ** argv) {116 bool is_run_by_cli = (argv == nullptr);117 118 common_models_handler models_handler;119 120 // note: router mode also accepts -hf remote-preset, so we need to check that first121 if (!is_run_by_cli && !params.model.hf_repo.empty()) {122 try {123 models_handler = common_models_handler_init(params, LLAMA_EXAMPLE_SERVER);124 if (common_models_handler_is_preset_repo(models_handler)) {125 // apply the preset and start the server in router mode126 common_models_handler_apply(models_handler, params);127 }128 } catch (const std::exception & e) {129 SRV_ERR("failed to fetch model metadata: %s\n", e.what());130 return 1;131 }132 }133 134 // router server never loads a model and must not touch the GPU135 const bool is_router_server = params.model.path.empty()136 && params.model.hf_repo.empty();137 138 // skip device enumeration so the CUDA primary context stays uncreated139 common_params_print_info(params, !is_router_server);140 141 if (!is_router_server) {142 // validate batch size for embeddings143 // embeddings require all tokens to be processed in a single ubatch144 // see https://github.com/ggml-org/llama.cpp/issues/12836145 if (params.embedding && params.n_batch > params.n_ubatch) {146 SRV_WRN("embeddings enabled with n_batch (%d) > n_ubatch (%d)\n", params.n_batch, params.n_ubatch);147 SRV_WRN("setting n_batch = n_ubatch = %d to avoid assertion failure\n", params.n_ubatch);148 params.n_batch = params.n_ubatch;149 }150 151 if (params.n_parallel < 0) {152 SRV_TRC("%s", "n_parallel is set to auto, using n_parallel = 4 and kv_unified = true\n");153 154 params.n_parallel = 4;155 params.kv_unified = true;156 }157 }158 159 // for consistency between server router mode and single-model mode, we set the same model name as alias160 auto model_name = params.model.get_name();161 if (params.model_alias.empty() && !model_name.empty()) {162 params.model_alias.insert(model_name);163 }164 165 // note: this is guaranteed to out-live ctx_http and tools166 server_mcp mcp_mgr;167 168 // struct that contains llama context and inference169 server_context ctx_server;170 171 server_http_context ctx_http;172 if (!ctx_http.init(params)) {173 SRV_ERR("%s", "failed to initialize HTTP server\n");174 return 1;175 }176 177 //178 // Router179 //180 181 // register API routes182 server_child child; // only used in non-router mode183 server_routes routes(params, ctx_server);184 server_tools tools;185 186 std::optional<server_models_routes> models_routes{};187 if (is_router_server) {188 // setup server instances manager189 try {190 models_routes.emplace(params, argc, argv);191 } catch (const std::exception & e) {192 SRV_ERR("failed to initialize router models: %s\n", e.what());193 return 1;194 }195 196 // proxy handlers197 // note: routes.get_health stays the same198 routes.get_metrics = models_routes->proxy_get;199 routes.post_props = models_routes->proxy_post;200 routes.post_completions = models_routes->proxy_post;201 routes.post_completions_oai = models_routes->proxy_post;202 routes.post_chat_completions = models_routes->proxy_post;203 routes.post_control = models_routes->proxy_post;204 routes.post_responses_oai = models_routes->proxy_post;205 routes.post_transcriptions_oai = models_routes->proxy_post;206 routes.post_anthropic_messages = models_routes->proxy_post;207 routes.post_anthropic_count_tokens = models_routes->proxy_post;208 routes.post_infill = models_routes->proxy_post;209 routes.post_embeddings = models_routes->proxy_post;210 routes.post_embeddings_oai = models_routes->proxy_post;211 routes.post_rerank = models_routes->proxy_post;212 routes.post_tokenize = models_routes->proxy_post;213 routes.post_detokenize = models_routes->proxy_post;214 routes.post_apply_template = models_routes->proxy_post;215 routes.post_chat_completions_tok = models_routes->proxy_post;216 routes.post_responses_tok_oai = models_routes->proxy_post;217 routes.get_lora_adapters = models_routes->proxy_get;218 routes.post_lora_adapters = models_routes->proxy_post;219 routes.get_slots = models_routes->proxy_get;220 routes.post_slots = models_routes->proxy_post;221 222 // custom routes for router223 routes.get_props = models_routes->get_router_props;224 routes.get_models = models_routes->get_router_models;225 226 ctx_http.post("/models", ex_wrapper(models_routes->post_router_models));227 ctx_http.post("/models/load", ex_wrapper(models_routes->post_router_models_load));228 ctx_http.post("/models/unload", ex_wrapper(models_routes->post_router_models_unload));229 ctx_http.get ("/models/sse", ex_wrapper(models_routes->get_router_models_sse));230 ctx_http.del ("/models", ex_wrapper(models_routes->del_router_models));231 }232 233 ctx_http.get ("/health", ex_wrapper(routes.get_health)); // public endpoint (no API key check)234 ctx_http.get ("/v1/health", ex_wrapper(routes.get_health)); // public endpoint (no API key check)235 ctx_http.get ("/metrics", ex_wrapper(routes.get_metrics));236 ctx_http.get ("/props", ex_wrapper(routes.get_props));237 ctx_http.post("/props", ex_wrapper(routes.post_props));238 ctx_http.get ("/models", ex_wrapper(routes.get_models)); // public endpoint (no API key check)239 ctx_http.get ("/v1/models", ex_wrapper(routes.get_models)); // public endpoint (no API key check)240 ctx_http.post("/completion", ex_wrapper(routes.post_completions)); // legacy241 ctx_http.post("/completions", ex_wrapper(routes.post_completions));242 ctx_http.post("/v1/completions", ex_wrapper(routes.post_completions_oai));243 ctx_http.post("/chat/completions", ex_wrapper(routes.post_chat_completions));244 ctx_http.post("/v1/chat/completions", ex_wrapper(routes.post_chat_completions));245 ctx_http.post("/v1/chat/completions/control", ex_wrapper(routes.post_control));246 ctx_http.post("/v1/responses", ex_wrapper(routes.post_responses_oai));247 ctx_http.post("/responses", ex_wrapper(routes.post_responses_oai));248 ctx_http.post("/v1/audio/transcriptions", ex_wrapper(routes.post_transcriptions_oai));249 ctx_http.post("/audio/transcriptions", ex_wrapper(routes.post_transcriptions_oai));250 ctx_http.post("/v1/messages", ex_wrapper(routes.post_anthropic_messages)); // anthropic messages API251 ctx_http.post("/infill", ex_wrapper(routes.post_infill));252 ctx_http.post("/embedding", ex_wrapper(routes.post_embeddings)); // legacy253 ctx_http.post("/embeddings", ex_wrapper(routes.post_embeddings));254 ctx_http.post("/v1/embeddings", ex_wrapper(routes.post_embeddings_oai));255 ctx_http.post("/rerank", ex_wrapper(routes.post_rerank));256 ctx_http.post("/reranking", ex_wrapper(routes.post_rerank));257 ctx_http.post("/v1/rerank", ex_wrapper(routes.post_rerank));258 ctx_http.post("/v1/reranking", ex_wrapper(routes.post_rerank));259 ctx_http.post("/tokenize", ex_wrapper(routes.post_tokenize));260 ctx_http.post("/detokenize", ex_wrapper(routes.post_detokenize));261 ctx_http.post("/apply-template", ex_wrapper(routes.post_apply_template));262 // token counting263 ctx_http.post("/chat/completions/input_tokens", ex_wrapper(routes.post_chat_completions_tok));264 ctx_http.post("/v1/chat/completions/input_tokens", ex_wrapper(routes.post_chat_completions_tok));265 ctx_http.post("/responses/input_tokens", ex_wrapper(routes.post_responses_tok_oai));266 ctx_http.post("/v1/responses/input_tokens", ex_wrapper(routes.post_responses_tok_oai));267 ctx_http.post("/v1/messages/count_tokens", ex_wrapper(routes.post_anthropic_count_tokens)); // anthropic token counting268 // LoRA adapters hotswap269 ctx_http.get ("/lora-adapters", ex_wrapper(routes.get_lora_adapters));270 ctx_http.post("/lora-adapters", ex_wrapper(routes.post_lora_adapters));271 // Save & load slots272 ctx_http.get ("/slots", ex_wrapper(routes.get_slots));273 ctx_http.post("/slots/:id_slot", ex_wrapper(routes.post_slots));274 275 // resumable streaming: a child binds the local session factories, the router binds276 // proxies that resolve the owning child, see server-stream.h277 server_http_context::handler_t stream_get_h;278 server_http_context::handler_t streams_lookup_h;279 server_http_context::handler_t stream_delete_h;280 if (is_router_server) {281 stream_get_h = models_routes->router_stream_get;282 streams_lookup_h = models_routes->router_streams_lookup;283 stream_delete_h = models_routes->router_stream_delete;284 } else {285 stream_get_h = server_stream_make_get_handler();286 streams_lookup_h = server_stream_make_lookup_handler();287 stream_delete_h = server_stream_make_delete_handler();288 }289 ctx_http.get ("/v1/stream", ex_wrapper(stream_get_h));290 ctx_http.post("/v1/streams/lookup", ex_wrapper(streams_lookup_h));291 ctx_http.del ("/v1/stream", ex_wrapper(stream_delete_h));292 293 // Google Cloud Platform (Vertex AI) compat294 ctx_http.register_gcp_compat();295 296 // return 403 for disabled features297 server_http_context::handler_t res_403 = [](const server_http_req &) {298 auto res = std::make_unique<server_http_res>();299 res->status = 403;300 res->data = safe_json_to_str({301 {"error", {302 {"message", "this feature is disabled"},303 {"type", "feature_disabled"},304 }}305 });306 return res;307 };308 309 if (params.cors_origins == "*" && params.api_keys.empty()) {310 SRV_WRN("%s", "-----------------\n");311 SRV_WRN("%s", "CORS is set to allow all origins ('*') and no API key is set\n");312 SRV_WRN("%s", "this can be a security risk (cross-origin attacks)\n");313 SRV_WRN("%s", "more info: https://github.com/ggml-org/llama.cpp/pull/25655\n");314 SRV_WRN("%s", "-----------------\n");315 }316 317 // CORS proxy (EXPERIMENTAL, only used by the Web UI for MCP)318 std::vector<std::string> warn_names;319 if (is_router_server) {320 warn_names.push_back("router mode");321 }322 323 if (params.ui_mcp_proxy) {324 ctx_http.get ("/cors-proxy", ex_wrapper(proxy_handler_get));325 ctx_http.post("/cors-proxy", ex_wrapper(proxy_handler_post));326 warn_names.push_back("MCP proxy (experimental)");327 } else {328 ctx_http.get ("/cors-proxy", ex_wrapper(res_403));329 ctx_http.post("/cors-proxy", ex_wrapper(res_403));330 }331 332 try {333 mcp_mgr.start(params);334 } catch (const std::exception & e) {335 SRV_ERR("MCP starting failed: %s\n", e.what());336 return 1;337 }338 339 if (!params.server_tools.empty() || !mcp_mgr.empty()) {340 try {341 tools.setup(params.server_tools, mcp_mgr, params.server_tools_runtime);342 } catch (const std::exception & e) {343 SRV_ERR("tools setup failed: %s\n", e.what());344 return 1;345 }346 ctx_http.get ("/tools", ex_wrapper(tools.handle_get));347 ctx_http.post("/tools", ex_wrapper(tools.handle_post));348 if (!params.server_tools.empty()) {349 warn_names.push_back("built-in tools (experimental)");350 }351 if (!params.server_tools_runtime.empty()) {352 warn_names.push_back("tools runtime (experimental)");353 }354 if (!mcp_mgr.empty()) {355 warn_names.push_back("MCP servers (experimental)");356 }357 } else {358 ctx_http.get ("/tools", ex_wrapper(res_403));359 ctx_http.post("/tools", ex_wrapper(res_403));360 }361 362 if (warn_names.size() > 0) {363 SRV_WRN("%s", "-----------------\n");364 SRV_WRN("%s", "the following feature(s) are enabled:\n");365 for (const auto & name : warn_names) {366 SRV_WRN(" %s\n", name.c_str());367 }368 SRV_WRN("%s", "do not expose the server to untrusted environments\n");369 SRV_WRN("%s", "-----------------\n");370 }371 372 //373 // Handle downloading model374 //375 376 if (child.is_child() && child.get_mode() == SERVER_CHILD_MODE_DOWNLOAD) {377 return child.run_download(params);378 } else if (!is_router_server && !is_run_by_cli) {379 // single-model mode (NOT spawned by router)380 // if this is invoked by CLI, model downloading should be already handled381 try {382 common_models_handler_apply(models_handler, params);383 } catch (const std::exception & e) {384 SRV_ERR("failed to download model: %s\n", e.what());385 return 1;386 }387 }388 389 //390 // Start the server391 //392 393 std::function<void()> clean_up;394 395 if (is_router_server) {396 SRV_INF("%s", "starting server in router mode. models will be automatically loaded on-demand\n");397 398 clean_up = [&models_routes, &mcp_mgr]() {399 SRV_INF("%s: cleaning up before exit...\n", __func__);400 // stop the session GC first, it finalizes live sessions and wakes pending readers401 server_stream_session_manager_stop();402 if (models_routes.has_value()) {403 models_routes->stopping.store(true); // maybe redundant, but just to be safe404 models_routes->models.unload_all();405 }406 mcp_mgr.shutdown();407 llama_backend_free();408 };409 410 if (!ctx_http.start()) {411 clean_up();412 SRV_ERR("%s", "exiting due to HTTP server error\n");413 return 1;414 }415 ctx_http.is_ready.store(true);416 417 shutdown_handler = [&](int) {418 if (models_routes.has_value()) {419 // important to disconnect any SSE clients420 models_routes->stopping.store(true);421 }422 mcp_mgr.shutdown();423 ctx_http.stop();424 };425 426 } else {427 // setup clean up function, to be called before exit428 clean_up = [&ctx_http, &ctx_server, &mcp_mgr]() {429 SRV_INF("%s: cleaning up before exit...\n", __func__);430 // stop the session GC first, it finalizes live sessions and wakes pending readers431 server_stream_session_manager_stop();432 ctx_http.stop();433 ctx_server.terminate();434 mcp_mgr.shutdown();435 llama_backend_free();436 };437 438 // start the HTTP server before loading the model to be able to serve /health requests439 if (!ctx_http.start()) {440 clean_up();441 SRV_ERR("%s", "exiting due to HTTP server error\n");442 return 1;443 }444 445 // setup communication child --> router if necessary446 if (child.is_child()) {447 ctx_server.set_state_callback([&](server_state state, json payload) {448 child.notify_to_router(server_state_to_str(state), payload);449 });450 }451 452 if (!ctx_server.load_model(params)) {453 clean_up();454 if (ctx_http.thread.joinable()) {455 ctx_http.thread.join();456 }457 SRV_ERR("%s", "exiting due to model loading error\n");458 return 1;459 }460 461 routes.update_meta(ctx_server);462 ctx_http.is_ready.store(true);463 464 SRV_INF("%s", "model loaded\n");465 466 shutdown_handler = [&](int) {467 mcp_mgr.shutdown();468 // this will unblock start_loop()469 ctx_server.terminate();470 };471 }472 473 // register signal handler if not running by CLI474 if (!is_run_by_cli) {475#if defined (__unix__) || (defined (__APPLE__) && defined (__MACH__))476 struct sigaction sigint_action;477 sigint_action.sa_handler = signal_handler;478 sigemptyset (&sigint_action.sa_mask);479 sigint_action.sa_flags = 0;480 sigaction(SIGINT, &sigint_action, NULL);481 sigaction(SIGTERM, &sigint_action, NULL);482#elif defined (_WIN32)483 auto console_ctrl_handler = +[](DWORD ctrl_type) -> BOOL {484 return (ctrl_type == CTRL_C_EVENT) ? (signal_handler(SIGINT), true) : false;485 };486 SetConsoleCtrlHandler(reinterpret_cast<PHANDLER_ROUTINE>(console_ctrl_handler), true);487#endif488 }489 490 SRV_INF("listening on %s\n", ctx_http.listening_address.c_str());491 492 // TODO: remove this in the future493 // check the string to also handle the .sock case494 if (string_ends_with(ctx_http.listening_address, ":8080")) {495 SRV_WRN("%s", "NOTICE: server default port will be changed to :9931 in a future release\n");496 SRV_WRN("%s", " ref: https://github.com/ggml-org/llama.cpp/pull/26508\n");497 }498 499 if (is_router_server) {500 if (!params.models_preset_hf.empty()) {501 SRV_WRN( "NOTE: using preset.ini from HF repo '%s'\n", params.models_preset_hf.c_str());502 SRV_WRN("%s", " please only use presets that you can trust! Unknown presets may be unsafe\n");503 }504 505 if (ctx_http.thread.joinable()) {506 ctx_http.thread.join(); // keep the main thread alive507 }508 509 // when the HTTP server stops, clean up and exit510 clean_up();511 } else {512 // optionally, notify router server that this instance is ready513 std::thread monitor_thread;514 if (child.is_child()) {515 monitor_thread = child.setup(shutdown_handler);516 child.notify_to_router(server_state_to_str(SERVER_STATE_READY), routes.get_model_info());517 }518 519 // this call blocks the main thread until queue_tasks.terminate() is called520 ctx_server.start_loop();521 522 clean_up();523 if (ctx_http.thread.joinable()) {524 ctx_http.thread.join();525 }526 if (monitor_thread.joinable()) {527 monitor_thread.join();528 }529 530 auto * ll_ctx = ctx_server.get_llama_context();531 if (ll_ctx != nullptr) {532 common_memory_breakdown_print(ll_ctx);533 }534 }535 536 return 0;537}538 