Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#include "server-task.h"2 3#include "build-info.h"4#include "server-chat.h"5#include "chat.h"6#include "common.h"7#include "json-schema-to-grammar.h"8#include "llama.h"9#include "sampling.h"10#include "speculative.h"11#include "server-common.h"12 13using json = nlohmann::ordered_json;14 15//16// task_params17//18 19json task_params::format_logit_bias(const std::vector<llama_logit_bias> & logit_bias) const {20 json data = json::array();21 for (const auto & lb : logit_bias) {22 data.push_back(json{23 {"bias", lb.bias},24 {"token", lb.token},25 });26 }27 return data;28}29 30json task_params::to_json(bool only_metrics) const {31 std::vector<std::string> samplers;32 samplers.reserve(sampling.samplers.size());33 for (const auto & sampler : sampling.samplers) {34 samplers.emplace_back(common_sampler_type_to_str(sampler));35 }36 37 json lora = json::array();38 for (auto & it : this->lora) {39 lora.push_back({{"id", it.first}, {"scale", it.second}});40 }41 42 if (only_metrics) {43 return json {44 {"seed", sampling.seed},45 {"temperature", sampling.temp},46 {"dynatemp_range", sampling.dynatemp_range},47 {"dynatemp_exponent", sampling.dynatemp_exponent},48 {"top_k", sampling.top_k},49 {"top_p", sampling.top_p},50 {"min_p", sampling.min_p},51 {"top_n_sigma", sampling.top_n_sigma},52 {"xtc_probability", sampling.xtc_probability},53 {"xtc_threshold", sampling.xtc_threshold},54 {"typical_p", sampling.typ_p},55 {"repeat_last_n", sampling.penalty_last_n},56 {"repeat_penalty", sampling.penalty_repeat},57 {"presence_penalty", sampling.penalty_present},58 {"frequency_penalty", sampling.penalty_freq},59 {"dry_multiplier", sampling.dry_multiplier},60 {"dry_base", sampling.dry_base},61 {"dry_allowed_length", sampling.dry_allowed_length},62 {"dry_penalty_last_n", sampling.dry_penalty_last_n},63 {"mirostat", sampling.mirostat},64 {"mirostat_tau", sampling.mirostat_tau},65 {"mirostat_eta", sampling.mirostat_eta},66 {"adaptive_target", sampling.adaptive_target},67 {"adaptive_decay", sampling.adaptive_decay},68 {"max_tokens", n_predict},69 {"n_predict", n_predict}, // TODO: deduplicate?70 {"n_keep", n_keep},71 {"n_discard", n_discard},72 {"ignore_eos", sampling.ignore_eos},73 {"stream", stream},74 {"n_probs", sampling.n_probs},75 {"min_keep", sampling.min_keep},76 {"chat_format", common_chat_format_name(chat_parser_params.format)},77 {"reasoning_format", common_reasoning_format_name(chat_parser_params.reasoning_format)},78 {"reasoning_in_content", chat_parser_params.reasoning_in_content},79 {"generation_prompt", chat_parser_params.generation_prompt},80 {"samplers", samplers},81 {"speculative.types", common_speculative_type_name_str(speculative.types)},82 {"timings_per_token", timings_per_token},83 {"post_sampling_probs", post_sampling_probs},84 {"backend_sampling", sampling.backend_sampling},85 {"lora", lora},86 };87 }88 89 auto grammar_triggers = json::array();90 for (const auto & trigger : sampling.grammar_triggers) {91 server_grammar_trigger ct(trigger);92 grammar_triggers.push_back(ct.to_json());93 }94 95 return json {96 {"seed", sampling.seed},97 {"temperature", sampling.temp},98 {"dynatemp_range", sampling.dynatemp_range},99 {"dynatemp_exponent", sampling.dynatemp_exponent},100 {"top_k", sampling.top_k},101 {"top_p", sampling.top_p},102 {"min_p", sampling.min_p},103 {"top_n_sigma", sampling.top_n_sigma},104 {"xtc_probability", sampling.xtc_probability},105 {"xtc_threshold", sampling.xtc_threshold},106 {"typical_p", sampling.typ_p},107 {"repeat_last_n", sampling.penalty_last_n},108 {"repeat_penalty", sampling.penalty_repeat},109 {"presence_penalty", sampling.penalty_present},110 {"frequency_penalty", sampling.penalty_freq},111 {"dry_multiplier", sampling.dry_multiplier},112 {"dry_base", sampling.dry_base},113 {"dry_allowed_length", sampling.dry_allowed_length},114 {"dry_penalty_last_n", sampling.dry_penalty_last_n},115 {"dry_sequence_breakers", sampling.dry_sequence_breakers},116 {"mirostat", sampling.mirostat},117 {"mirostat_tau", sampling.mirostat_tau},118 {"mirostat_eta", sampling.mirostat_eta},119 {"adaptive_target", sampling.adaptive_target},120 {"adaptive_decay", sampling.adaptive_decay},121 {"stop", antiprompt},122 {"max_tokens", n_predict},123 {"n_predict", n_predict}, // TODO: deduplicate?124 {"n_keep", n_keep},125 {"n_discard", n_discard},126 {"ignore_eos", sampling.ignore_eos},127 {"stream", stream},128 {"logit_bias", format_logit_bias(sampling.logit_bias)},129 {"n_probs", sampling.n_probs},130 {"min_keep", sampling.min_keep},131 {"grammar", common_grammar_value(sampling.grammar)},132 {"grammar_lazy", sampling.grammar_lazy},133 {"grammar_triggers", grammar_triggers},134 {"preserved_tokens", sampling.preserved_tokens},135 {"chat_format", common_chat_format_name(chat_parser_params.format)},136 {"reasoning_format", common_reasoning_format_name(chat_parser_params.reasoning_format)},137 {"reasoning_in_content", chat_parser_params.reasoning_in_content},138 {"generation_prompt", chat_parser_params.generation_prompt},139 {"samplers", samplers},140 {"speculative.types", common_speculative_type_name_str(speculative.types)},141 {"timings_per_token", timings_per_token},142 {"post_sampling_probs", post_sampling_probs},143 {"backend_sampling", sampling.backend_sampling},144 {"lora", lora},145 };146}147 148//149// task_result_state150//151task_result_state::task_result_state(const common_chat_parser_params & chat_parser_params)152 : chat_parser_params(chat_parser_params)153 , oai_resp_id("resp_" + random_string())154 , oai_resp_reasoning_id("rs_" + random_string())155 , oai_resp_message_id("msg_" + random_string()) {156 if (chat_parser_params.is_continuation && !chat_parser_params.echo) {157 // initialize chat_msg to avoid emitting a delta containing the assistant prefill158 chat_msg = common_chat_parse("", true, chat_parser_params);159 }160}161 162common_chat_msg task_result_state::update_chat_msg(163 const std::string & text_added,164 bool is_partial,165 std::vector<common_chat_msg_diff> & diffs,166 bool filter_tool_calls) {167 generated_text += text_added;168 auto msg_prv_copy = chat_msg;169 //SRV_DBG("Parsing chat message: %s\n", generated_text.c_str());170 auto new_msg = common_chat_parse(171 generated_text,172 is_partial,173 chat_parser_params);174 if (!new_msg.empty()) {175 new_msg.set_tool_call_ids(generated_tool_call_ids, gen_tool_call_id);176 chat_msg = new_msg;177 auto all_diffs = common_chat_msg_diff::compute_diffs(msg_prv_copy, chat_msg);178 179 if (!filter_tool_calls) {180 diffs = std::move(all_diffs);181 } else {182 for (auto & d : all_diffs) {183 // If this is a new type of delta, flush all currently pending tool call names184 for (size_t i = 0; i < chat_msg.tool_calls.size(); ++i) {185 if (sent_tool_call_names.count(i) || chat_msg.tool_calls[i].name.empty()) {186 continue;187 }188 if (d.tool_call_index != i || !d.tool_call_delta.arguments.empty()) {189 common_chat_msg_diff header;190 header.tool_call_index = i;191 header.tool_call_delta.id = chat_msg.tool_calls[i].id;192 header.tool_call_delta.name = chat_msg.tool_calls[i].name;193 diffs.push_back(std::move(header));194 sent_tool_call_names.insert(i);195 }196 }197 198 if (d.tool_call_index == std::string::npos) {199 diffs.push_back(std::move(d));200 } else {201 size_t i = d.tool_call_index;202 if (sent_tool_call_names.count(i)) {203 if (!d.tool_call_delta.arguments.empty()) {204 d.tool_call_delta.name = "";205 d.tool_call_delta.id = "";206 diffs.push_back(std::move(d));207 }208 } else {209 // Not sent yet.210 if (!d.tool_call_delta.arguments.empty() || !is_partial) {211 d.tool_call_delta.name = chat_msg.tool_calls[i].name;212 d.tool_call_delta.id = chat_msg.tool_calls[i].id;213 diffs.push_back(std::move(d));214 sent_tool_call_names.insert(i);215 } else {216 // Suppress217 }218 }219 }220 }221 // Final check at EOF222 if (!is_partial) {223 for (size_t i = 0; i < chat_msg.tool_calls.size(); ++i) {224 if (!sent_tool_call_names.count(i) && !chat_msg.tool_calls[i].name.empty()) {225 common_chat_msg_diff header;226 header.tool_call_index = i;227 header.tool_call_delta.id = chat_msg.tool_calls[i].id;228 header.tool_call_delta.name = chat_msg.tool_calls[i].name;229 diffs.push_back(std::move(header));230 sent_tool_call_names.insert(i);231 }232 }233 }234 }235 }236 return chat_msg;237}238 239//240 241// result_timings242//243 244json result_timings::to_json() const {245 json base = {246 {"cache_n", cache_n},247 248 {"prompt_n", prompt_n},249 {"prompt_ms", prompt_ms},250 {"prompt_per_token_ms", prompt_per_token_ms},251 {"prompt_per_second", prompt_per_second},252 253 {"predicted_n", predicted_n},254 {"predicted_ms", predicted_ms},255 {"predicted_per_token_ms", predicted_per_token_ms},256 {"predicted_per_second", predicted_per_second},257 };258 259 if (draft_n > 0) {260 base["draft_n"] = draft_n;261 base["draft_n_accepted"] = draft_n_accepted;262 }263 264 return base;265}266 267//268// result_prompt_progress269//270json result_prompt_progress::to_json() const {271 return json {272 {"total", total},273 {"cache", cache},274 {"processed", processed},275 {"time_ms", time_ms},276 };277}278 279static inline std::string stop_type_to_str(stop_type type) {280 switch (type) {281 case STOP_TYPE_EOS: return "eos";282 case STOP_TYPE_WORD: return "word";283 case STOP_TYPE_LIMIT: return "limit";284 default: return "none";285 }286}287 288//289// completion_token_output290//291 292json completion_token_output::to_json(bool post_sampling_probs) const {293 json probs_for_token = json::array();294 for (const auto & p : probs) {295 std::string txt(p.txt);296 txt.resize(validate_utf8(txt));297 probs_for_token.push_back(json {298 {"id", p.tok},299 {"token", txt},300 {"bytes", str_to_bytes(p.txt)},301 {302 post_sampling_probs ? "prob" : "logprob",303 post_sampling_probs ? p.prob : logarithm(p.prob)304 },305 });306 }307 return probs_for_token;308}309 310json completion_token_output::probs_vector_to_json(const std::vector<completion_token_output> & probs, bool post_sampling_probs) {311 json out = json::array();312 for (const auto & p : probs) {313 std::string txt(p.text_to_send);314 txt.resize(validate_utf8(txt));315 out.push_back(json {316 {"id", p.tok},317 {"token", txt},318 {"bytes", str_to_bytes(p.text_to_send)},319 {320 post_sampling_probs ? "prob" : "logprob",321 post_sampling_probs ? p.prob : logarithm(p.prob)322 },323 {324 post_sampling_probs ? "top_probs" : "top_logprobs",325 p.to_json(post_sampling_probs)326 },327 });328 }329 return out;330}331 332float completion_token_output::logarithm(float x) {333 // nlohmann::json converts -inf to null, so we need to prevent that334 return x == 0.0f ? std::numeric_limits<float>::lowest() : std::log(x);335}336 337std::vector<unsigned char> completion_token_output::str_to_bytes(const std::string & str) {338 std::vector<unsigned char> bytes;339 for (unsigned char c : str) {340 bytes.push_back(c);341 }342 return bytes;343}344 345//346// server_task_result_cmpl_final347//348json server_task_result_cmpl_final::to_json() {349 GGML_ASSERT(is_updated && "update() must be called before to_json()");350 switch (res_type) {351 case TASK_RESPONSE_TYPE_NONE:352 return to_json_non_oaicompat();353 case TASK_RESPONSE_TYPE_OAI_CMPL:354 return to_json_oaicompat();355 case TASK_RESPONSE_TYPE_OAI_CHAT:356 return stream ? to_json_oaicompat_chat_stream() : to_json_oaicompat_chat();357 case TASK_RESPONSE_TYPE_OAI_RESP:358 return stream ? to_json_oaicompat_resp_stream() : to_json_oaicompat_resp();359 case TASK_RESPONSE_TYPE_OAI_ASR:360 return to_json_oaicompat_asr();361 case TASK_RESPONSE_TYPE_ANTHROPIC:362 return stream ? to_json_anthropic_stream() : to_json_anthropic();363 default:364 GGML_ASSERT(false && "Invalid task_response_type");365 }366}367 368json server_task_result_cmpl_final::to_json_non_oaicompat() {369 json res = json {370 {"index", index},371 {"content", content},372 {"tokens", tokens},373 {"id_slot", id_slot},374 {"stop", true},375 {"model", oaicompat_model},376 {"tokens_predicted", n_decoded},377 {"tokens_evaluated", n_prompt_tokens},378 {"generation_settings", generation_params.to_json()},379 {"prompt", prompt},380 {"has_new_line", has_new_line},381 {"truncated", truncated},382 {"stop_type", stop_type_to_str(stop)},383 {"stopping_word", stopping_word},384 {"tokens_cached", n_tokens_cached},385 {"timings", timings.to_json()},386 };387 if (!stream && !probs_output.empty()) {388 res["completion_probabilities"] = completion_token_output::probs_vector_to_json(probs_output, post_sampling_probs);389 }390 return response_fields.empty() ? res : json_get_nested_values(response_fields, res);391}392 393json server_task_result_cmpl_final::usage_json_oaicompat() {394 return json {395 {"completion_tokens", n_decoded},396 {"prompt_tokens", n_prompt_tokens},397 {"total_tokens", n_decoded + n_prompt_tokens},398 {"prompt_tokens_details", json { {"cached_tokens", n_prompt_tokens_cache} }},399 };400}401 402json server_task_result_cmpl_final::to_json_oaicompat() {403 std::time_t t = std::time(0);404 json logprobs = json(nullptr); // OAI default to null405 if (!stream && probs_output.size() > 0) {406 logprobs = json{407 {"content", completion_token_output::probs_vector_to_json(probs_output, post_sampling_probs)},408 };409 }410 json finish_reason = "length";411 if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {412 finish_reason = "stop";413 }414 json res = json {415 {"choices", json::array({416 json{417 {"text", content},418 {"index", index},419 {"logprobs", logprobs},420 {"finish_reason", finish_reason},421 }422 })},423 {"created", t},424 {"model", oaicompat_model},425 {"system_fingerprint", std::string(llama_build_info())},426 {"object", "text_completion"},427 {"usage", usage_json_oaicompat()},428 {"id", oaicompat_cmpl_id}429 };430 431 // extra fields for debugging purposes432 if (verbose) {433 res["__verbose"] = to_json_non_oaicompat();434 }435 if (timings.prompt_n >= 0) {436 res.push_back({"timings", timings.to_json()});437 }438 439 return res;440}441 442json server_task_result_cmpl_final::to_json_oaicompat_chat() {443 std::string finish_reason = "length";444 common_chat_msg msg;445 if (!oaicompat_msg.empty()) {446 msg = oaicompat_msg;447 } else {448 msg.role = "assistant";449 msg.content = content;450 }451 if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {452 finish_reason = msg.tool_calls.empty() ? "stop" : "tool_calls";453 }454 455 json choice {456 {"finish_reason", finish_reason},457 {"index", index},458 {"message", msg.to_json_oaicompat()},459 };460 461 if (!stream && probs_output.size() > 0) {462 choice["logprobs"] = json{463 {"content", completion_token_output::probs_vector_to_json(probs_output, post_sampling_probs)},464 };465 }466 467 std::time_t t = std::time(0);468 469 json res = json {470 {"choices", json::array({choice})},471 {"created", t},472 {"model", oaicompat_model},473 {"system_fingerprint", std::string(llama_build_info())},474 {"object", "chat.completion"},475 {"usage", usage_json_oaicompat()},476 {"id", oaicompat_cmpl_id}477 };478 479 // extra fields for debugging purposes480 if (verbose) {481 res["__verbose"] = to_json_non_oaicompat();482 }483 if (timings.prompt_n >= 0) {484 res.push_back({"timings", timings.to_json()});485 }486 487 return res;488}489 490json server_task_result_cmpl_final::to_json_oaicompat_chat_stream() {491 std::time_t t = std::time(0);492 std::string finish_reason = "length";493 if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {494 finish_reason = oaicompat_msg.tool_calls.empty() ? "stop" : "tool_calls";495 }496 497 json deltas = json::array();498 for (const auto & diff : oaicompat_msg_diffs) {499 deltas.push_back({500 {"choices", json::array({501 json {502 {"finish_reason", nullptr},503 {"index", index},504 {"delta", server_chat_msg_diff_to_json_oaicompat(diff)},505 },506 })},507 {"created", t},508 {"id", oaicompat_cmpl_id},509 {"model", oaicompat_model},510 {"system_fingerprint", std::string(llama_build_info())},511 {"object", "chat.completion.chunk"},512 });513 }514 515 deltas.push_back({516 {"choices", json::array({517 json {518 {"finish_reason", finish_reason},519 {"index", index},520 {"delta", json::object()},521 },522 })},523 {"created", t},524 {"id", oaicompat_cmpl_id},525 {"model", oaicompat_model},526 {"system_fingerprint", std::string(llama_build_info())},527 {"object", "chat.completion.chunk"},528 });529 530 if (include_usage) {531 // OpenAI API spec for chat.completion.chunks specifies an empty `choices` array for the last chunk when including usage532 // https://platform.openai.com/docs/api-reference/chat_streaming/streaming#chat_streaming/streaming-choices533 deltas.push_back({534 {"choices", json::array()},535 {"created", t},536 {"id", oaicompat_cmpl_id},537 {"model", oaicompat_model},538 {"system_fingerprint", std::string(llama_build_info())},539 {"object", "chat.completion.chunk"},540 {"usage", usage_json_oaicompat()},541 });542 }543 544 if (timings.prompt_n >= 0) {545 deltas.back().push_back({"timings", timings.to_json()});546 }547 548 // extra fields for debugging purposes549 if (verbose && !deltas.empty()) {550 deltas.front()["__verbose"] = to_json_non_oaicompat();551 }552 553 return deltas;554}555 556json server_task_result_cmpl_final::to_json_oaicompat_resp() {557 common_chat_msg msg;558 if (!oaicompat_msg.empty()) {559 msg = oaicompat_msg;560 } else {561 msg.role = "assistant";562 msg.content = content;563 }564 565 std::vector<json> output;566 567 if (msg.reasoning_content != "") {568 output.push_back(json {569 {"id", "rs_" + random_string()},570 {"summary", json::array()},571 {"type", "reasoning"},572 {"content", json::array({ json {573 {"text", msg.reasoning_content},574 {"type", "reasoning_text"},575 }})},576 {"encrypted_content", ""},577 {"status", "completed"},578 });579 }580 581 if (msg.content != "") {582 output.push_back(json {583 {"content", json::array({ json {584 {"type", "output_text"},585 {"annotations", json::array()},586 {"logprobs", json::array()},587 {"text", msg.content},588 }})},589 {"id", "msg_" + random_string()},590 {"role", msg.role},591 {"status", "completed"},592 {"type", "message"},593 });594 }595 596 for (const common_chat_tool_call & tool_call : oaicompat_msg.tool_calls) {597 output.push_back(json {598 {"id", "fc_" + tool_call.id},599 {"type", "function_call"},600 {"status", "completed"},601 {"arguments", tool_call.arguments},602 {"call_id", "call_" + tool_call.id},603 {"name", tool_call.name},604 });605 }606 607 std::time_t t = std::time(0);608 json res = {609 {"completed_at", t},610 {"created_at", t},611 {"id", oai_resp_id},612 {"model", oaicompat_model},613 {"object", "response"},614 {"output", output},615 {"status", "completed"},616 {"usage", json {617 {"input_tokens", n_prompt_tokens},618 {"output_tokens", n_decoded},619 {"total_tokens", n_decoded + n_prompt_tokens},620 {"input_tokens_details", json { {"cached_tokens", n_prompt_tokens_cache} }},621 }},622 };623 624 return res;625}626 627json server_task_result_cmpl_final::to_json_oaicompat_resp_stream() {628 std::vector<json> server_sent_events;629 std::vector<json> output;630 631 if (oaicompat_msg.reasoning_content != "") {632 const json output_item = json {633 {"id", oai_resp_reasoning_id},634 {"summary", json::array()},635 {"type", "reasoning"},636 {"content", json::array({ json {637 {"text", oaicompat_msg.reasoning_content},638 {"type", "reasoning_text"},639 }})},640 {"encrypted_content", ""},641 };642 643 server_sent_events.push_back(json {644 {"event", "response.output_item.done"},645 {"data", json {646 {"type", "response.output_item.done"},647 {"item", output_item}648 }}649 });650 output.push_back(output_item);651 }652 653 if (oaicompat_msg.content != "") {654 server_sent_events.push_back(json {655 {"event", "response.output_text.done"},656 {"data", json {657 {"type", "response.output_text.done"},658 {"item_id", oai_resp_message_id},659 {"text", oaicompat_msg.content}660 }}661 });662 663 const json content_part = {664 {"type", "output_text"},665 {"annotations", json::array()},666 {"logprobs", json::array()},667 {"text", oaicompat_msg.content}668 };669 670 server_sent_events.push_back(json {671 {"event", "response.content_part.done"},672 {"data", json {673 {"type", "response.content_part.done"},674 {"item_id", oai_resp_message_id},675 {"part", content_part}676 }}677 });678 const json output_item = {679 {"type", "message"},680 {"status", "completed"},681 {"id", oai_resp_message_id},682 {"content", json::array({content_part})},683 {"role", "assistant"}684 };685 686 server_sent_events.push_back(json {687 {"event", "response.output_item.done"},688 {"data", json {689 {"type", "response.output_item.done"},690 {"item", output_item}691 }}692 });693 output.push_back(output_item);694 }695 696 for (const common_chat_tool_call & tool_call : oaicompat_msg.tool_calls) {697 const json output_item = {698 {"id", "fc_" + tool_call.id},699 {"type", "function_call"},700 {"status", "completed"},701 {"arguments", tool_call.arguments},702 {"call_id", "call_" + tool_call.id},703 {"name", tool_call.name}704 };705 server_sent_events.push_back(json {706 {"event", "response.output_item.done"},707 {"data", json {708 {"type", "response.output_item.done"},709 {"item", output_item}710 }}711 });712 output.push_back(output_item);713 }714 715 std::time_t t = std::time(0);716 server_sent_events.push_back(json {717 {"event", "response.completed"},718 {"data", json {719 {"type", "response.completed"},720 {"response", json {721 {"id", oai_resp_id},722 {"object", "response"},723 {"created_at", t},724 {"status", "completed"},725 {"model", oaicompat_model},726 {"output", output},727 {"usage", json {728 {"input_tokens", n_prompt_tokens},729 {"output_tokens", n_decoded},730 {"total_tokens", n_decoded + n_prompt_tokens},731 {"input_tokens_details", json { {"cached_tokens", n_prompt_tokens_cache} }},732 }}733 }},734 }}735 });736 737 if (timings.prompt_n >= 0) {738 server_sent_events.back().at("data").push_back({"timings", timings.to_json()});739 }740 741 return server_sent_events;742}743 744json server_task_result_cmpl_final::to_json_oaicompat_asr() {745 json event = json {746 {"type", "transcript.text.done"},747 {"text", oaicompat_msg.content},748 {"usage", json {749 {"type", "tokens"},750 {"input_tokens", n_prompt_tokens},751 {"output_tokens", n_decoded},752 {"total_tokens", n_decoded + n_prompt_tokens},753 {"input_tokens_details", json { {"cached_tokens", n_prompt_tokens_cache} }},754 }},755 };756 return event;757}758 759json server_task_result_cmpl_final::to_json_anthropic() {760 std::string stop_reason = "max_tokens";761 if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {762 stop_reason = oaicompat_msg.tool_calls.empty() ? "end_turn" : "tool_use";763 }764 765 json content_blocks = json::array();766 767 common_chat_msg msg;768 if (!oaicompat_msg.empty()) {769 msg = oaicompat_msg;770 } else {771 msg.role = "assistant";772 msg.content = content;773 }774 775 // thinking block comes first (Anthropic extended thinking format)776 if (!msg.reasoning_content.empty()) {777 content_blocks.push_back({778 {"type", "thinking"},779 {"thinking", msg.reasoning_content},780 {"signature", ""} // empty signature for local models (no cryptographic verification)781 });782 }783 784 if (!msg.content.empty()) {785 content_blocks.push_back({786 {"type", "text"},787 {"text", msg.content}788 });789 }790 791 for (const auto & tool_call : msg.tool_calls) {792 json tool_use_block = {793 {"type", "tool_use"},794 {"id", tool_call.id},795 {"name", tool_call.name}796 };797 798 try {799 tool_use_block["input"] = json::parse(tool_call.arguments);800 } catch (const std::exception &) {801 tool_use_block["input"] = json::object();802 }803 804 content_blocks.push_back(tool_use_block);805 }806 807 json res = {808 {"id", oaicompat_cmpl_id},809 {"type", "message"},810 {"role", "assistant"},811 {"content", content_blocks},812 {"model", oaicompat_model},813 {"stop_reason", stop_reason},814 {"stop_sequence", stopping_word.empty() ? nullptr : json(stopping_word)},815 {"usage", {816 {"cache_read_input_tokens", n_prompt_tokens_cache},817 {"input_tokens", n_prompt_tokens - n_prompt_tokens_cache},818 {"output_tokens", n_decoded}819 }}820 };821 822 return res;823}824 825json server_task_result_cmpl_final::to_json_anthropic_stream() {826 json events = json::array();827 828 std::string stop_reason = "max_tokens";829 if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {830 stop_reason = oaicompat_msg.tool_calls.empty() ? "end_turn" : "tool_use";831 }832 833 bool has_thinking = !oaicompat_msg.reasoning_content.empty();834 bool has_text = !oaicompat_msg.content.empty();835 size_t num_tool_calls = oaicompat_msg.tool_calls.size();836 837 // content block indices: thinking (0) -> text (0 or 1) -> tool_use (n+)838 size_t thinking_block_index = 0;839 size_t text_block_index = has_thinking ? 1 : 0;840 841 bool thinking_block_started = false;842 bool text_block_started = false;843 std::unordered_set<size_t> tool_calls_started;844 845 for (const auto & diff : oaicompat_msg_diffs) {846 // handle thinking/reasoning content847 if (!diff.reasoning_content_delta.empty()) {848 if (!thinking_block_started) {849 events.push_back({850 {"event", "content_block_start"},851 {"data", {852 {"type", "content_block_start"},853 {"index", thinking_block_index},854 {"content_block", {855 {"type", "thinking"},856 {"thinking", ""}857 }}858 }}859 });860 thinking_block_started = true;861 }862 863 events.push_back({864 {"event", "content_block_delta"},865 {"data", {866 {"type", "content_block_delta"},867 {"index", thinking_block_index},868 {"delta", {869 {"type", "thinking_delta"},870 {"thinking", diff.reasoning_content_delta}871 }}872 }}873 });874 }875 876 // handle regular text content877 if (!diff.content_delta.empty()) {878 if (!text_block_started) {879 events.push_back({880 {"event", "content_block_start"},881 {"data", {882 {"type", "content_block_start"},883 {"index", text_block_index},884 {"content_block", {885 {"type", "text"},886 {"text", ""}887 }}888 }}889 });890 text_block_started = true;891 }892 893 events.push_back({894 {"event", "content_block_delta"},895 {"data", {896 {"type", "content_block_delta"},897 {"index", text_block_index},898 {"delta", {899 {"type", "text_delta"},900 {"text", diff.content_delta}901 }}902 }}903 });904 }905 906 // handle tool calls907 if (diff.tool_call_index != std::string::npos) {908 size_t content_block_index = (has_thinking ? 1 : 0) + (has_text ? 1 : 0) + diff.tool_call_index;909 910 if (tool_calls_started.find(diff.tool_call_index) == tool_calls_started.end()) {911 const auto & full_tool_call = oaicompat_msg.tool_calls[diff.tool_call_index];912 913 events.push_back({914 {"event", "content_block_start"},915 {"data", {916 {"type", "content_block_start"},917 {"index", content_block_index},918 {"content_block", {919 {"type", "tool_use"},920 {"id", full_tool_call.id},921 {"name", full_tool_call.name}922 }}923 }}924 });925 tool_calls_started.insert(diff.tool_call_index);926 }927 928 if (!diff.tool_call_delta.arguments.empty()) {929 events.push_back({930 {"event", "content_block_delta"},931 {"data", {932 {"type", "content_block_delta"},933 {"index", content_block_index},934 {"delta", {935 {"type", "input_json_delta"},936 {"partial_json", diff.tool_call_delta.arguments}937 }}938 }}939 });940 }941 }942 }943 944 // close content blocks in order945 if (has_thinking) {946 // Anthropic API requires a signature_delta before closing thinking blocks947 // We use an empty signature since we can't generate a cryptographic signature for local models948 events.push_back({949 {"event", "content_block_delta"},950 {"data", {951 {"type", "content_block_delta"},952 {"index", thinking_block_index},953 {"delta", {954 {"type", "signature_delta"},955 {"signature", ""}956 }}957 }}958 });959 events.push_back({960 {"event", "content_block_stop"},961 {"data", {962 {"type", "content_block_stop"},963 {"index", thinking_block_index}964 }}965 });966 }967 968 if (has_text) {969 events.push_back({970 {"event", "content_block_stop"},971 {"data", {972 {"type", "content_block_stop"},973 {"index", text_block_index}974 }}975 });976 }977 978 for (size_t i = 0; i < num_tool_calls; i++) {979 size_t content_block_index = (has_thinking ? 1 : 0) + (has_text ? 1 : 0) + i;980 events.push_back({981 {"event", "content_block_stop"},982 {"data", {983 {"type", "content_block_stop"},984 {"index", content_block_index}985 }}986 });987 }988 989 events.push_back({990 {"event", "message_delta"},991 {"data", {992 {"type", "message_delta"},993 {"delta", {994 {"stop_reason", stop_reason},995 {"stop_sequence", stopping_word.empty() ? nullptr : json(stopping_word)}996 }},997 {"usage", {998 {"output_tokens", n_decoded}999 }}1000 }}1001 });1002 1003 events.push_back({1004 {"event", "message_stop"},1005 {"data", {1006 {"type", "message_stop"}1007 }}1008 });1009 1010 return events;1011}1012 1013//1014// server_task_result_cmpl_partial1015//1016void server_task_result_cmpl_partial::update(task_result_state & state) {1017 is_updated = true;1018 if (is_begin) {1019 return; // begin marker only flushes headers, skip parsing1020 }1021 state.update_chat_msg(content, true, oaicompat_msg_diffs);1022 1023 // Copy current state for use in to_json_*() (reflects state BEFORE this chunk)1024 thinking_block_started = state.thinking_block_started;1025 text_block_started = state.text_block_started;1026 1027 oai_resp_created = state.oai_resp_created;1028 oai_resp_id = state.oai_resp_id;1029 oai_resp_reasoning_id = state.oai_resp_reasoning_id;1030 oai_resp_message_id = state.oai_resp_message_id;1031 oai_resp_fc_id = state.oai_resp_fc_id;1032 1033 // track if the accumulated message has any reasoning content1034 anthropic_has_reasoning = !state.chat_msg.reasoning_content.empty();1035 1036 if (res_type == TASK_RESPONSE_TYPE_OAI_RESP && !state.oai_resp_created && (is_progress || n_decoded == 1)) {1037 state.oai_resp_created = true;1038 }1039 1040 // Pre-compute state updates based on diffs (for next chunk)1041 for (const common_chat_msg_diff & diff : oaicompat_msg_diffs) {1042 if (!diff.reasoning_content_delta.empty() && !state.thinking_block_started) {1043 state.thinking_block_started = true;1044 }1045 if (!diff.content_delta.empty() && !state.text_block_started) {1046 state.text_block_started = true;1047 }1048 if (!diff.tool_call_delta.name.empty()) {1049 state.oai_resp_fc_id = diff.tool_call_delta.id;1050 }1051 }1052}1053 1054json server_task_result_cmpl_partial::to_json() {1055 GGML_ASSERT(is_updated && "update() must be called before to_json()");1056 if (is_begin) {1057 return nullptr; // simply signal to HTTP handler to send the headers and status code1058 }1059 switch (res_type) {1060 case TASK_RESPONSE_TYPE_NONE:1061 return to_json_non_oaicompat();1062 case TASK_RESPONSE_TYPE_OAI_CMPL:1063 return to_json_oaicompat();1064 case TASK_RESPONSE_TYPE_OAI_CHAT:1065 return to_json_oaicompat_chat();1066 case TASK_RESPONSE_TYPE_OAI_RESP:1067 return to_json_oaicompat_resp();1068 case TASK_RESPONSE_TYPE_OAI_ASR:1069 return to_json_oaicompat_asr();1070 case TASK_RESPONSE_TYPE_ANTHROPIC:1071 return to_json_anthropic();1072 default:1073 GGML_ASSERT(false && "Invalid task_response_type");1074 }1075}1076 1077json server_task_result_cmpl_partial::to_json_non_oaicompat() {1078 // non-OAI-compat JSON1079 json res = json {1080 {"index", index},1081 {"content", content},1082 {"tokens", tokens},1083 {"stop", false},1084 {"id_slot", id_slot},1085 {"tokens_predicted", n_decoded},1086 {"tokens_evaluated", n_prompt_tokens},1087 };1088 // populate the timings object when needed (usually for the last response or with timings_per_token enabled)1089 if (timings.prompt_n > 0) {1090 res.push_back({"timings", timings.to_json()});1091 }1092 if (is_progress) {1093 res.push_back({"prompt_progress", progress.to_json()});1094 }1095 if (!prob_output.probs.empty()) {1096 res["completion_probabilities"] = completion_token_output::probs_vector_to_json({prob_output}, post_sampling_probs);1097 }1098 return res;1099}1100 1101json server_task_result_cmpl_partial::to_json_oaicompat() {1102 std::time_t t = std::time(0);1103 json logprobs = json(nullptr); // OAI default to null1104 if (prob_output.probs.size() > 0) {1105 logprobs = json{1106 {"content", completion_token_output::probs_vector_to_json({prob_output}, post_sampling_probs)},1107 };1108 }1109 json res = json {1110 {"choices", json::array({1111 json{1112 {"text", content},1113 {"index", index},1114 {"logprobs", logprobs},1115 {"finish_reason", nullptr},1116 }1117 })},1118 {"created", t},1119 {"model", oaicompat_model},1120 {"system_fingerprint", std::string(llama_build_info())},1121 {"object", "text_completion"},1122 {"id", oaicompat_cmpl_id}1123 };1124 1125 // extra fields for debugging purposes1126 if (verbose) {1127 res["__verbose"] = to_json_non_oaicompat();1128 }1129 if (timings.prompt_n >= 0) {1130 res.push_back({"timings", timings.to_json()});1131 }1132 if (is_progress) {1133 res.push_back({"prompt_progress", progress.to_json()});1134 }1135 1136 return res;1137}1138 1139json server_task_result_cmpl_partial::to_json_oaicompat_chat() {1140 bool first = n_decoded == 1;1141 std::time_t t = std::time(0);1142 json choices;1143 1144 std::vector<json> deltas;1145 auto add_delta = [&](const json & delta) {1146 deltas.push_back({1147 {"choices", json::array({1148 json {1149 {"finish_reason", nullptr},1150 {"index", index},1151 {"delta", delta},1152 },1153 })},1154 {"created", t},1155 {"id", oaicompat_cmpl_id},1156 {"model", oaicompat_model},1157 {"system_fingerprint", std::string(llama_build_info())},1158 {"object", "chat.completion.chunk"},1159 });1160 };1161 // We have to send an initial update to conform to openai behavior1162 if (first || is_progress) {1163 add_delta({1164 {"role", "assistant"},1165 {"content", nullptr},1166 });1167 }1168 1169 for (const auto & diff : oaicompat_msg_diffs) {1170 add_delta(server_chat_msg_diff_to_json_oaicompat(diff));1171 }1172 1173 if (!deltas.empty()) {1174 auto & last_json = deltas[deltas.size() - 1];1175 GGML_ASSERT(last_json.at("choices").size() >= 1);1176 1177 if (prob_output.probs.size() > 0) {1178 last_json.at("choices").at(0)["logprobs"] = json {1179 {"content", completion_token_output::probs_vector_to_json({prob_output}, post_sampling_probs)},1180 };1181 }1182 1183 if (timings.prompt_n >= 0) {1184 last_json.push_back({"timings", timings.to_json()});1185 }1186 if (is_progress) {1187 last_json.push_back({"prompt_progress", progress.to_json()});1188 }1189 }1190 1191 return deltas;1192}1193 1194json server_task_result_cmpl_partial::to_json_oaicompat_resp() {1195 std::vector<json> events;1196 1197 if (!oai_resp_created) {1198 events.push_back(json {1199 {"event", "response.created"},1200 {"data", json {