Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
server-task.cpp1855 linesDownload Raw Back to server
1#include "server-task.h"2 3#include "build-info.h"4#include "server-chat.h"5#include "chat.h"6#include "common.h"7#include "json-schema-to-grammar.h"8#include "llama.h"9#include "sampling.h"10#include "speculative.h"11#include "server-common.h"12 13using json = nlohmann::ordered_json;14 15//16// task_params17//18 19json task_params::format_logit_bias(const std::vector<llama_logit_bias> & logit_bias) const {20    json data = json::array();21    for (const auto & lb : logit_bias) {22        data.push_back(json{23            {"bias", lb.bias},24            {"token", lb.token},25        });26    }27    return data;28}29 30json task_params::to_json(bool only_metrics) const {31    std::vector<std::string> samplers;32    samplers.reserve(sampling.samplers.size());33    for (const auto & sampler : sampling.samplers) {34        samplers.emplace_back(common_sampler_type_to_str(sampler));35    }36 37    json lora = json::array();38    for (auto & it : this->lora) {39        lora.push_back({{"id", it.first}, {"scale", it.second}});40    }41 42    if (only_metrics) {43        return json {44            {"seed",                      sampling.seed},45            {"temperature",               sampling.temp},46            {"dynatemp_range",            sampling.dynatemp_range},47            {"dynatemp_exponent",         sampling.dynatemp_exponent},48            {"top_k",                     sampling.top_k},49            {"top_p",                     sampling.top_p},50            {"min_p",                     sampling.min_p},51            {"top_n_sigma",               sampling.top_n_sigma},52            {"xtc_probability",           sampling.xtc_probability},53            {"xtc_threshold",             sampling.xtc_threshold},54            {"typical_p",                 sampling.typ_p},55            {"repeat_last_n",             sampling.penalty_last_n},56            {"repeat_penalty",            sampling.penalty_repeat},57            {"presence_penalty",          sampling.penalty_present},58            {"frequency_penalty",         sampling.penalty_freq},59            {"dry_multiplier",            sampling.dry_multiplier},60            {"dry_base",                  sampling.dry_base},61            {"dry_allowed_length",        sampling.dry_allowed_length},62            {"dry_penalty_last_n",        sampling.dry_penalty_last_n},63            {"mirostat",                  sampling.mirostat},64            {"mirostat_tau",              sampling.mirostat_tau},65            {"mirostat_eta",              sampling.mirostat_eta},66            {"adaptive_target",           sampling.adaptive_target},67            {"adaptive_decay",            sampling.adaptive_decay},68            {"max_tokens",                n_predict},69            {"n_predict",                 n_predict}, // TODO: deduplicate?70            {"n_keep",                    n_keep},71            {"n_discard",                 n_discard},72            {"ignore_eos",                sampling.ignore_eos},73            {"stream",                    stream},74            {"n_probs",                   sampling.n_probs},75            {"min_keep",                  sampling.min_keep},76            {"chat_format",               common_chat_format_name(chat_parser_params.format)},77            {"reasoning_format",          common_reasoning_format_name(chat_parser_params.reasoning_format)},78            {"reasoning_in_content",      chat_parser_params.reasoning_in_content},79            {"generation_prompt",         chat_parser_params.generation_prompt},80            {"samplers",                  samplers},81            {"speculative.types",         common_speculative_type_name_str(speculative.types)},82            {"timings_per_token",         timings_per_token},83            {"post_sampling_probs",       post_sampling_probs},84            {"backend_sampling",          sampling.backend_sampling},85            {"lora",                      lora},86        };87    }88 89    auto grammar_triggers = json::array();90    for (const auto & trigger : sampling.grammar_triggers) {91        server_grammar_trigger ct(trigger);92        grammar_triggers.push_back(ct.to_json());93    }94 95    return json {96        {"seed",                      sampling.seed},97        {"temperature",               sampling.temp},98        {"dynatemp_range",            sampling.dynatemp_range},99        {"dynatemp_exponent",         sampling.dynatemp_exponent},100        {"top_k",                     sampling.top_k},101        {"top_p",                     sampling.top_p},102        {"min_p",                     sampling.min_p},103        {"top_n_sigma",               sampling.top_n_sigma},104        {"xtc_probability",           sampling.xtc_probability},105        {"xtc_threshold",             sampling.xtc_threshold},106        {"typical_p",                 sampling.typ_p},107        {"repeat_last_n",             sampling.penalty_last_n},108        {"repeat_penalty",            sampling.penalty_repeat},109        {"presence_penalty",          sampling.penalty_present},110        {"frequency_penalty",         sampling.penalty_freq},111        {"dry_multiplier",            sampling.dry_multiplier},112        {"dry_base",                  sampling.dry_base},113        {"dry_allowed_length",        sampling.dry_allowed_length},114        {"dry_penalty_last_n",        sampling.dry_penalty_last_n},115        {"dry_sequence_breakers",     sampling.dry_sequence_breakers},116        {"mirostat",                  sampling.mirostat},117        {"mirostat_tau",              sampling.mirostat_tau},118        {"mirostat_eta",              sampling.mirostat_eta},119        {"adaptive_target",           sampling.adaptive_target},120        {"adaptive_decay",            sampling.adaptive_decay},121        {"stop",                      antiprompt},122        {"max_tokens",                n_predict},123        {"n_predict",                 n_predict}, // TODO: deduplicate?124        {"n_keep",                    n_keep},125        {"n_discard",                 n_discard},126        {"ignore_eos",                sampling.ignore_eos},127        {"stream",                    stream},128        {"logit_bias",                format_logit_bias(sampling.logit_bias)},129        {"n_probs",                   sampling.n_probs},130        {"min_keep",                  sampling.min_keep},131        {"grammar",                   common_grammar_value(sampling.grammar)},132        {"grammar_lazy",              sampling.grammar_lazy},133        {"grammar_triggers",          grammar_triggers},134        {"preserved_tokens",          sampling.preserved_tokens},135        {"chat_format",               common_chat_format_name(chat_parser_params.format)},136        {"reasoning_format",          common_reasoning_format_name(chat_parser_params.reasoning_format)},137        {"reasoning_in_content",      chat_parser_params.reasoning_in_content},138        {"generation_prompt",         chat_parser_params.generation_prompt},139        {"samplers",                  samplers},140        {"speculative.types",         common_speculative_type_name_str(speculative.types)},141        {"timings_per_token",         timings_per_token},142        {"post_sampling_probs",       post_sampling_probs},143        {"backend_sampling",          sampling.backend_sampling},144        {"lora",                      lora},145    };146}147 148//149// task_result_state150//151task_result_state::task_result_state(const common_chat_parser_params & chat_parser_params)152    : chat_parser_params(chat_parser_params)153    , oai_resp_id("resp_" + random_string())154    , oai_resp_reasoning_id("rs_" + random_string())155    , oai_resp_message_id("msg_" + random_string()) {156    if (chat_parser_params.is_continuation && !chat_parser_params.echo) {157        // initialize chat_msg to avoid emitting a delta containing the assistant prefill158        chat_msg = common_chat_parse("", true, chat_parser_params);159    }160}161 162common_chat_msg task_result_state::update_chat_msg(163        const std::string & text_added,164        bool is_partial,165        std::vector<common_chat_msg_diff> & diffs,166        bool filter_tool_calls) {167    generated_text += text_added;168    auto msg_prv_copy = chat_msg;169    //SRV_DBG("Parsing chat message: %s\n", generated_text.c_str());170    auto new_msg = common_chat_parse(171        generated_text,172        is_partial,173        chat_parser_params);174    if (!new_msg.empty()) {175        new_msg.set_tool_call_ids(generated_tool_call_ids, gen_tool_call_id);176        chat_msg = new_msg;177        auto all_diffs = common_chat_msg_diff::compute_diffs(msg_prv_copy, chat_msg);178 179        if (!filter_tool_calls) {180            diffs = std::move(all_diffs);181        } else {182            for (auto & d : all_diffs) {183                // If this is a new type of delta, flush all currently pending tool call names184                for (size_t i = 0; i < chat_msg.tool_calls.size(); ++i) {185                    if (sent_tool_call_names.count(i) || chat_msg.tool_calls[i].name.empty()) {186                        continue;187                    }188                    if (d.tool_call_index != i || !d.tool_call_delta.arguments.empty()) {189                        common_chat_msg_diff header;190                        header.tool_call_index      = i;191                        header.tool_call_delta.id   = chat_msg.tool_calls[i].id;192                        header.tool_call_delta.name = chat_msg.tool_calls[i].name;193                        diffs.push_back(std::move(header));194                        sent_tool_call_names.insert(i);195                    }196                }197 198                if (d.tool_call_index == std::string::npos) {199                    diffs.push_back(std::move(d));200                } else {201                    size_t i = d.tool_call_index;202                    if (sent_tool_call_names.count(i)) {203                        if (!d.tool_call_delta.arguments.empty()) {204                            d.tool_call_delta.name = "";205                            d.tool_call_delta.id   = "";206                            diffs.push_back(std::move(d));207                        }208                    } else {209                        // Not sent yet.210                        if (!d.tool_call_delta.arguments.empty() || !is_partial) {211                            d.tool_call_delta.name = chat_msg.tool_calls[i].name;212                            d.tool_call_delta.id   = chat_msg.tool_calls[i].id;213                            diffs.push_back(std::move(d));214                            sent_tool_call_names.insert(i);215                        } else {216                            // Suppress217                        }218                    }219                }220            }221            // Final check at EOF222            if (!is_partial) {223                for (size_t i = 0; i < chat_msg.tool_calls.size(); ++i) {224                    if (!sent_tool_call_names.count(i) && !chat_msg.tool_calls[i].name.empty()) {225                        common_chat_msg_diff header;226                        header.tool_call_index      = i;227                        header.tool_call_delta.id   = chat_msg.tool_calls[i].id;228                        header.tool_call_delta.name = chat_msg.tool_calls[i].name;229                        diffs.push_back(std::move(header));230                        sent_tool_call_names.insert(i);231                    }232                }233            }234        }235    }236    return chat_msg;237}238 239//240 241// result_timings242//243 244json result_timings::to_json() const {245    json base = {246        {"cache_n",                cache_n},247 248        {"prompt_n",               prompt_n},249        {"prompt_ms",              prompt_ms},250        {"prompt_per_token_ms",    prompt_per_token_ms},251        {"prompt_per_second",      prompt_per_second},252 253        {"predicted_n",            predicted_n},254        {"predicted_ms",           predicted_ms},255        {"predicted_per_token_ms", predicted_per_token_ms},256        {"predicted_per_second",   predicted_per_second},257    };258 259    if (draft_n > 0) {260        base["draft_n"] = draft_n;261        base["draft_n_accepted"] = draft_n_accepted;262    }263 264    return base;265}266 267//268// result_prompt_progress269//270json result_prompt_progress::to_json() const {271    return json {272        {"total",     total},273        {"cache",     cache},274        {"processed", processed},275        {"time_ms",   time_ms},276    };277}278 279static inline std::string stop_type_to_str(stop_type type) {280    switch (type) {281        case STOP_TYPE_EOS:   return "eos";282        case STOP_TYPE_WORD:  return "word";283        case STOP_TYPE_LIMIT: return "limit";284        default:              return "none";285    }286}287 288//289// completion_token_output290//291 292json completion_token_output::to_json(bool post_sampling_probs) const {293    json probs_for_token = json::array();294    for (const auto & p : probs) {295        std::string txt(p.txt);296        txt.resize(validate_utf8(txt));297        probs_for_token.push_back(json {298            {"id",      p.tok},299            {"token",   txt},300            {"bytes",   str_to_bytes(p.txt)},301            {302                post_sampling_probs ? "prob" : "logprob",303                post_sampling_probs ? p.prob : logarithm(p.prob)304            },305        });306    }307    return probs_for_token;308}309 310json completion_token_output::probs_vector_to_json(const std::vector<completion_token_output> & probs, bool post_sampling_probs) {311    json out = json::array();312    for (const auto & p : probs) {313        std::string txt(p.text_to_send);314        txt.resize(validate_utf8(txt));315        out.push_back(json {316            {"id",           p.tok},317            {"token",        txt},318            {"bytes",        str_to_bytes(p.text_to_send)},319            {320                post_sampling_probs ? "prob" : "logprob",321                post_sampling_probs ? p.prob : logarithm(p.prob)322            },323            {324                post_sampling_probs ? "top_probs" : "top_logprobs",325                p.to_json(post_sampling_probs)326            },327        });328    }329    return out;330}331 332float completion_token_output::logarithm(float x) {333    // nlohmann::json converts -inf to null, so we need to prevent that334    return x == 0.0f ? std::numeric_limits<float>::lowest() : std::log(x);335}336 337std::vector<unsigned char> completion_token_output::str_to_bytes(const std::string & str) {338    std::vector<unsigned char> bytes;339    for (unsigned char c : str) {340        bytes.push_back(c);341    }342    return bytes;343}344 345//346// server_task_result_cmpl_final347//348json server_task_result_cmpl_final::to_json() {349    GGML_ASSERT(is_updated && "update() must be called before to_json()");350    switch (res_type) {351        case TASK_RESPONSE_TYPE_NONE:352            return to_json_non_oaicompat();353        case TASK_RESPONSE_TYPE_OAI_CMPL:354            return to_json_oaicompat();355        case TASK_RESPONSE_TYPE_OAI_CHAT:356            return stream ? to_json_oaicompat_chat_stream() : to_json_oaicompat_chat();357        case TASK_RESPONSE_TYPE_OAI_RESP:358            return stream ? to_json_oaicompat_resp_stream() : to_json_oaicompat_resp();359        case TASK_RESPONSE_TYPE_OAI_ASR:360            return to_json_oaicompat_asr();361        case TASK_RESPONSE_TYPE_ANTHROPIC:362            return stream ? to_json_anthropic_stream() : to_json_anthropic();363        default:364            GGML_ASSERT(false && "Invalid task_response_type");365    }366}367 368json server_task_result_cmpl_final::to_json_non_oaicompat() {369    json res = json {370        {"index",               index},371        {"content",             content},372        {"tokens",              tokens},373        {"id_slot",             id_slot},374        {"stop",                true},375        {"model",               oaicompat_model},376        {"tokens_predicted",    n_decoded},377        {"tokens_evaluated",    n_prompt_tokens},378        {"generation_settings", generation_params.to_json()},379        {"prompt",              prompt},380        {"has_new_line",        has_new_line},381        {"truncated",           truncated},382        {"stop_type",           stop_type_to_str(stop)},383        {"stopping_word",       stopping_word},384        {"tokens_cached",       n_tokens_cached},385        {"timings",             timings.to_json()},386    };387    if (!stream && !probs_output.empty()) {388        res["completion_probabilities"] = completion_token_output::probs_vector_to_json(probs_output, post_sampling_probs);389    }390    return response_fields.empty() ? res : json_get_nested_values(response_fields, res);391}392 393json server_task_result_cmpl_final::usage_json_oaicompat() {394    return json {395        {"completion_tokens", n_decoded},396        {"prompt_tokens",     n_prompt_tokens},397        {"total_tokens",      n_decoded + n_prompt_tokens},398        {"prompt_tokens_details", json { {"cached_tokens", n_prompt_tokens_cache} }},399    };400}401 402json server_task_result_cmpl_final::to_json_oaicompat() {403    std::time_t t = std::time(0);404    json logprobs = json(nullptr); // OAI default to null405    if (!stream && probs_output.size() > 0) {406        logprobs = json{407            {"content", completion_token_output::probs_vector_to_json(probs_output, post_sampling_probs)},408        };409    }410    json finish_reason = "length";411    if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {412        finish_reason = "stop";413    }414    json res = json {415        {"choices",            json::array({416            json{417                {"text",          content},418                {"index",         index},419                {"logprobs",      logprobs},420                {"finish_reason", finish_reason},421            }422        })},423        {"created",            t},424        {"model",              oaicompat_model},425        {"system_fingerprint", std::string(llama_build_info())},426        {"object",             "text_completion"},427        {"usage",              usage_json_oaicompat()},428        {"id", oaicompat_cmpl_id}429    };430 431    // extra fields for debugging purposes432    if (verbose) {433        res["__verbose"] = to_json_non_oaicompat();434    }435    if (timings.prompt_n >= 0) {436        res.push_back({"timings", timings.to_json()});437    }438 439    return res;440}441 442json server_task_result_cmpl_final::to_json_oaicompat_chat() {443    std::string finish_reason = "length";444    common_chat_msg msg;445    if (!oaicompat_msg.empty()) {446        msg = oaicompat_msg;447    } else {448        msg.role = "assistant";449        msg.content = content;450    }451    if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {452        finish_reason = msg.tool_calls.empty() ? "stop" : "tool_calls";453    }454 455    json choice {456        {"finish_reason", finish_reason},457        {"index", index},458        {"message", msg.to_json_oaicompat()},459    };460 461    if (!stream && probs_output.size() > 0) {462        choice["logprobs"] = json{463            {"content", completion_token_output::probs_vector_to_json(probs_output, post_sampling_probs)},464        };465    }466 467    std::time_t t = std::time(0);468 469    json res = json {470        {"choices",            json::array({choice})},471        {"created",            t},472        {"model",              oaicompat_model},473        {"system_fingerprint", std::string(llama_build_info())},474        {"object",             "chat.completion"},475        {"usage",              usage_json_oaicompat()},476        {"id", oaicompat_cmpl_id}477    };478 479    // extra fields for debugging purposes480    if (verbose) {481        res["__verbose"] = to_json_non_oaicompat();482    }483    if (timings.prompt_n >= 0) {484        res.push_back({"timings", timings.to_json()});485    }486 487    return res;488}489 490json server_task_result_cmpl_final::to_json_oaicompat_chat_stream() {491    std::time_t t = std::time(0);492    std::string finish_reason = "length";493    if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {494        finish_reason = oaicompat_msg.tool_calls.empty() ? "stop" : "tool_calls";495    }496 497    json deltas = json::array();498    for (const auto & diff : oaicompat_msg_diffs) {499        deltas.push_back({500            {"choices", json::array({501                json {502                    {"finish_reason", nullptr},503                    {"index", index},504                    {"delta", server_chat_msg_diff_to_json_oaicompat(diff)},505                },506            })},507            {"created", t},508            {"id", oaicompat_cmpl_id},509            {"model", oaicompat_model},510            {"system_fingerprint", std::string(llama_build_info())},511            {"object", "chat.completion.chunk"},512        });513    }514 515    deltas.push_back({516        {"choices", json::array({517            json {518                {"finish_reason", finish_reason},519                {"index", index},520                {"delta", json::object()},521            },522        })},523        {"created",            t},524        {"id",                 oaicompat_cmpl_id},525        {"model",              oaicompat_model},526        {"system_fingerprint", std::string(llama_build_info())},527        {"object",             "chat.completion.chunk"},528    });529 530    if (include_usage) {531        // OpenAI API spec for chat.completion.chunks specifies an empty `choices` array for the last chunk when including usage532        // https://platform.openai.com/docs/api-reference/chat_streaming/streaming#chat_streaming/streaming-choices533        deltas.push_back({534            {"choices", json::array()},535            {"created",            t},536            {"id",                 oaicompat_cmpl_id},537            {"model",              oaicompat_model},538            {"system_fingerprint", std::string(llama_build_info())},539            {"object",             "chat.completion.chunk"},540            {"usage",              usage_json_oaicompat()},541        });542    }543 544    if (timings.prompt_n >= 0) {545        deltas.back().push_back({"timings", timings.to_json()});546    }547 548    // extra fields for debugging purposes549    if (verbose && !deltas.empty()) {550        deltas.front()["__verbose"] = to_json_non_oaicompat();551    }552 553    return deltas;554}555 556json server_task_result_cmpl_final::to_json_oaicompat_resp() {557    common_chat_msg msg;558    if (!oaicompat_msg.empty()) {559        msg = oaicompat_msg;560    } else {561        msg.role = "assistant";562        msg.content = content;563    }564 565    std::vector<json> output;566 567    if (msg.reasoning_content != "") {568        output.push_back(json {569            {"id",      "rs_" + random_string()},570            {"summary", json::array()},571            {"type",    "reasoning"},572            {"content", json::array({ json {573                {"text", msg.reasoning_content},574                {"type", "reasoning_text"},575            }})},576            {"encrypted_content", ""},577            {"status",            "completed"},578        });579    }580 581    if (msg.content != "") {582        output.push_back(json {583            {"content", json::array({ json {584                {"type",        "output_text"},585                {"annotations", json::array()},586                {"logprobs",    json::array()},587                {"text",        msg.content},588            }})},589            {"id",     "msg_" + random_string()},590            {"role",   msg.role},591            {"status", "completed"},592            {"type",   "message"},593        });594    }595 596    for (const common_chat_tool_call & tool_call : oaicompat_msg.tool_calls) {597        output.push_back(json {598            {"id",        "fc_" + tool_call.id},599            {"type",      "function_call"},600            {"status",    "completed"},601            {"arguments", tool_call.arguments},602            {"call_id",   "call_" + tool_call.id},603            {"name",      tool_call.name},604        });605    }606 607    std::time_t t = std::time(0);608    json res = {609        {"completed_at", t},610        {"created_at",   t},611        {"id",           oai_resp_id},612        {"model",        oaicompat_model},613        {"object",       "response"},614        {"output",       output},615        {"status",       "completed"},616        {"usage",        json {617            {"input_tokens",  n_prompt_tokens},618            {"output_tokens", n_decoded},619            {"total_tokens",  n_decoded + n_prompt_tokens},620            {"input_tokens_details", json { {"cached_tokens", n_prompt_tokens_cache} }},621        }},622    };623 624    return res;625}626 627json server_task_result_cmpl_final::to_json_oaicompat_resp_stream() {628    std::vector<json> server_sent_events;629    std::vector<json> output;630 631    if (oaicompat_msg.reasoning_content != "") {632        const json output_item = json {633            {"id",      oai_resp_reasoning_id},634            {"summary", json::array()},635            {"type",    "reasoning"},636            {"content", json::array({ json {637                {"text", oaicompat_msg.reasoning_content},638                {"type", "reasoning_text"},639            }})},640            {"encrypted_content", ""},641        };642 643        server_sent_events.push_back(json {644            {"event", "response.output_item.done"},645            {"data", json {646                {"type", "response.output_item.done"},647                {"item", output_item}648            }}649        });650        output.push_back(output_item);651    }652 653    if (oaicompat_msg.content != "") {654        server_sent_events.push_back(json {655            {"event", "response.output_text.done"},656            {"data", json {657                {"type",    "response.output_text.done"},658                {"item_id", oai_resp_message_id},659                {"text",    oaicompat_msg.content}660            }}661        });662 663        const json content_part = {664            {"type",        "output_text"},665            {"annotations", json::array()},666            {"logprobs",    json::array()},667            {"text",        oaicompat_msg.content}668        };669 670        server_sent_events.push_back(json {671            {"event", "response.content_part.done"},672            {"data", json {673                {"type",    "response.content_part.done"},674                {"item_id", oai_resp_message_id},675                {"part",    content_part}676            }}677        });678        const json output_item = {679            {"type",    "message"},680            {"status",  "completed"},681            {"id",      oai_resp_message_id},682            {"content", json::array({content_part})},683            {"role",    "assistant"}684        };685 686        server_sent_events.push_back(json {687            {"event", "response.output_item.done"},688            {"data", json {689                {"type", "response.output_item.done"},690                {"item", output_item}691            }}692        });693        output.push_back(output_item);694    }695 696    for (const common_chat_tool_call & tool_call : oaicompat_msg.tool_calls) {697        const json output_item = {698            {"id",        "fc_" + tool_call.id},699            {"type",      "function_call"},700            {"status",    "completed"},701            {"arguments", tool_call.arguments},702            {"call_id",   "call_" + tool_call.id},703            {"name",      tool_call.name}704        };705        server_sent_events.push_back(json {706            {"event", "response.output_item.done"},707            {"data", json {708                {"type", "response.output_item.done"},709                {"item", output_item}710            }}711        });712        output.push_back(output_item);713    }714 715    std::time_t t = std::time(0);716    server_sent_events.push_back(json {717        {"event", "response.completed"},718        {"data", json {719            {"type", "response.completed"},720            {"response", json {721                {"id",         oai_resp_id},722                {"object",     "response"},723                {"created_at", t},724                {"status",     "completed"},725                {"model",      oaicompat_model},726                {"output",     output},727                {"usage",      json {728                    {"input_tokens",  n_prompt_tokens},729                    {"output_tokens", n_decoded},730                    {"total_tokens",  n_decoded + n_prompt_tokens},731                    {"input_tokens_details", json { {"cached_tokens", n_prompt_tokens_cache} }},732                }}733            }},734        }}735    });736 737    if (timings.prompt_n >= 0) {738        server_sent_events.back().at("data").push_back({"timings", timings.to_json()});739    }740 741    return server_sent_events;742}743 744json server_task_result_cmpl_final::to_json_oaicompat_asr() {745    json event = json {746        {"type",  "transcript.text.done"},747        {"text",  oaicompat_msg.content},748        {"usage", json {749            {"type",         "tokens"},750            {"input_tokens",  n_prompt_tokens},751            {"output_tokens", n_decoded},752            {"total_tokens",  n_decoded + n_prompt_tokens},753            {"input_tokens_details", json { {"cached_tokens", n_prompt_tokens_cache} }},754        }},755    };756    return event;757}758 759json server_task_result_cmpl_final::to_json_anthropic() {760    std::string stop_reason = "max_tokens";761    if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {762        stop_reason = oaicompat_msg.tool_calls.empty() ? "end_turn" : "tool_use";763    }764 765    json content_blocks = json::array();766 767    common_chat_msg msg;768    if (!oaicompat_msg.empty()) {769        msg = oaicompat_msg;770    } else {771        msg.role = "assistant";772        msg.content = content;773    }774 775    // thinking block comes first (Anthropic extended thinking format)776    if (!msg.reasoning_content.empty()) {777        content_blocks.push_back({778            {"type", "thinking"},779            {"thinking", msg.reasoning_content},780            {"signature", ""}  // empty signature for local models (no cryptographic verification)781        });782    }783 784    if (!msg.content.empty()) {785        content_blocks.push_back({786            {"type", "text"},787            {"text", msg.content}788        });789    }790 791    for (const auto & tool_call : msg.tool_calls) {792        json tool_use_block = {793            {"type", "tool_use"},794            {"id", tool_call.id},795            {"name", tool_call.name}796        };797 798        try {799            tool_use_block["input"] = json::parse(tool_call.arguments);800        } catch (const std::exception &) {801            tool_use_block["input"] = json::object();802        }803 804        content_blocks.push_back(tool_use_block);805    }806 807    json res = {808        {"id", oaicompat_cmpl_id},809        {"type", "message"},810        {"role", "assistant"},811        {"content", content_blocks},812        {"model", oaicompat_model},813        {"stop_reason", stop_reason},814        {"stop_sequence", stopping_word.empty() ? nullptr : json(stopping_word)},815        {"usage", {816            {"cache_read_input_tokens", n_prompt_tokens_cache},817            {"input_tokens", n_prompt_tokens - n_prompt_tokens_cache},818            {"output_tokens", n_decoded}819        }}820    };821 822    return res;823}824 825json server_task_result_cmpl_final::to_json_anthropic_stream() {826    json events = json::array();827 828    std::string stop_reason = "max_tokens";829    if (stop == STOP_TYPE_WORD || stop == STOP_TYPE_EOS) {830        stop_reason = oaicompat_msg.tool_calls.empty() ? "end_turn" : "tool_use";831    }832 833    bool has_thinking = !oaicompat_msg.reasoning_content.empty();834    bool has_text     = !oaicompat_msg.content.empty();835    size_t num_tool_calls = oaicompat_msg.tool_calls.size();836 837    // content block indices: thinking (0) -> text (0 or 1) -> tool_use (n+)838    size_t thinking_block_index = 0;839    size_t text_block_index     = has_thinking ? 1 : 0;840 841    bool thinking_block_started = false;842    bool text_block_started     = false;843    std::unordered_set<size_t> tool_calls_started;844 845    for (const auto & diff : oaicompat_msg_diffs) {846        // handle thinking/reasoning content847        if (!diff.reasoning_content_delta.empty()) {848            if (!thinking_block_started) {849                events.push_back({850                    {"event", "content_block_start"},851                    {"data", {852                        {"type", "content_block_start"},853                        {"index", thinking_block_index},854                        {"content_block", {855                            {"type", "thinking"},856                            {"thinking", ""}857                        }}858                    }}859                });860                thinking_block_started = true;861            }862 863            events.push_back({864                {"event", "content_block_delta"},865                {"data", {866                    {"type", "content_block_delta"},867                    {"index", thinking_block_index},868                    {"delta", {869                        {"type", "thinking_delta"},870                        {"thinking", diff.reasoning_content_delta}871                    }}872                }}873            });874        }875 876        // handle regular text content877        if (!diff.content_delta.empty()) {878            if (!text_block_started) {879                events.push_back({880                    {"event", "content_block_start"},881                    {"data", {882                        {"type", "content_block_start"},883                        {"index", text_block_index},884                        {"content_block", {885                            {"type", "text"},886                            {"text", ""}887                        }}888                    }}889                });890                text_block_started = true;891            }892 893            events.push_back({894                {"event", "content_block_delta"},895                {"data", {896                    {"type", "content_block_delta"},897                    {"index", text_block_index},898                    {"delta", {899                        {"type", "text_delta"},900                        {"text", diff.content_delta}901                    }}902                }}903            });904        }905 906        // handle tool calls907        if (diff.tool_call_index != std::string::npos) {908            size_t content_block_index = (has_thinking ? 1 : 0) + (has_text ? 1 : 0) + diff.tool_call_index;909 910            if (tool_calls_started.find(diff.tool_call_index) == tool_calls_started.end()) {911                const auto & full_tool_call = oaicompat_msg.tool_calls[diff.tool_call_index];912 913                events.push_back({914                    {"event", "content_block_start"},915                    {"data", {916                        {"type", "content_block_start"},917                        {"index", content_block_index},918                        {"content_block", {919                            {"type", "tool_use"},920                            {"id", full_tool_call.id},921                            {"name", full_tool_call.name}922                        }}923                    }}924                });925                tool_calls_started.insert(diff.tool_call_index);926            }927 928            if (!diff.tool_call_delta.arguments.empty()) {929                events.push_back({930                    {"event", "content_block_delta"},931                    {"data", {932                        {"type", "content_block_delta"},933                        {"index", content_block_index},934                        {"delta", {935                            {"type", "input_json_delta"},936                            {"partial_json", diff.tool_call_delta.arguments}937                        }}938                    }}939                });940            }941        }942    }943 944    // close content blocks in order945    if (has_thinking) {946        // Anthropic API requires a signature_delta before closing thinking blocks947        // We use an empty signature since we can't generate a cryptographic signature for local models948        events.push_back({949            {"event", "content_block_delta"},950            {"data", {951                {"type", "content_block_delta"},952                {"index", thinking_block_index},953                {"delta", {954                    {"type", "signature_delta"},955                    {"signature", ""}956                }}957            }}958        });959        events.push_back({960            {"event", "content_block_stop"},961            {"data", {962                {"type", "content_block_stop"},963                {"index", thinking_block_index}964            }}965        });966    }967 968    if (has_text) {969        events.push_back({970            {"event", "content_block_stop"},971            {"data", {972                {"type", "content_block_stop"},973                {"index", text_block_index}974            }}975        });976    }977 978    for (size_t i = 0; i < num_tool_calls; i++) {979        size_t content_block_index = (has_thinking ? 1 : 0) + (has_text ? 1 : 0) + i;980        events.push_back({981            {"event", "content_block_stop"},982            {"data", {983                {"type", "content_block_stop"},984                {"index", content_block_index}985            }}986        });987    }988 989    events.push_back({990        {"event", "message_delta"},991        {"data", {992            {"type", "message_delta"},993            {"delta", {994                {"stop_reason", stop_reason},995                {"stop_sequence", stopping_word.empty() ? nullptr : json(stopping_word)}996            }},997            {"usage", {998                {"output_tokens", n_decoded}999            }}1000        }}1001    });1002 1003    events.push_back({1004        {"event", "message_stop"},1005        {"data", {1006            {"type", "message_stop"}1007        }}1008    });1009 1010    return events;1011}1012 1013//1014// server_task_result_cmpl_partial1015//1016void server_task_result_cmpl_partial::update(task_result_state & state) {1017    is_updated = true;1018    if (is_begin) {1019        return; // begin marker only flushes headers, skip parsing1020    }1021    state.update_chat_msg(content, true, oaicompat_msg_diffs);1022 1023    // Copy current state for use in to_json_*() (reflects state BEFORE this chunk)1024    thinking_block_started = state.thinking_block_started;1025    text_block_started     = state.text_block_started;1026 1027    oai_resp_created       = state.oai_resp_created;1028    oai_resp_id            = state.oai_resp_id;1029    oai_resp_reasoning_id  = state.oai_resp_reasoning_id;1030    oai_resp_message_id    = state.oai_resp_message_id;1031    oai_resp_fc_id         = state.oai_resp_fc_id;1032 1033    // track if the accumulated message has any reasoning content1034    anthropic_has_reasoning = !state.chat_msg.reasoning_content.empty();1035 1036    if (res_type == TASK_RESPONSE_TYPE_OAI_RESP && !state.oai_resp_created && (is_progress || n_decoded == 1)) {1037        state.oai_resp_created = true;1038    }1039 1040    // Pre-compute state updates based on diffs (for next chunk)1041    for (const common_chat_msg_diff & diff : oaicompat_msg_diffs) {1042        if (!diff.reasoning_content_delta.empty() && !state.thinking_block_started) {1043            state.thinking_block_started = true;1044        }1045        if (!diff.content_delta.empty() && !state.text_block_started) {1046            state.text_block_started = true;1047        }1048        if (!diff.tool_call_delta.name.empty()) {1049            state.oai_resp_fc_id = diff.tool_call_delta.id;1050        }1051    }1052}1053 1054json server_task_result_cmpl_partial::to_json() {1055    GGML_ASSERT(is_updated && "update() must be called before to_json()");1056    if (is_begin) {1057        return nullptr; // simply signal to HTTP handler to send the headers and status code1058    }1059    switch (res_type) {1060        case TASK_RESPONSE_TYPE_NONE:1061            return to_json_non_oaicompat();1062        case TASK_RESPONSE_TYPE_OAI_CMPL:1063            return to_json_oaicompat();1064        case TASK_RESPONSE_TYPE_OAI_CHAT:1065            return to_json_oaicompat_chat();1066        case TASK_RESPONSE_TYPE_OAI_RESP:1067            return to_json_oaicompat_resp();1068        case TASK_RESPONSE_TYPE_OAI_ASR:1069            return to_json_oaicompat_asr();1070        case TASK_RESPONSE_TYPE_ANTHROPIC:1071            return to_json_anthropic();1072        default:1073            GGML_ASSERT(false && "Invalid task_response_type");1074    }1075}1076 1077json server_task_result_cmpl_partial::to_json_non_oaicompat() {1078    // non-OAI-compat JSON1079    json res = json {1080        {"index",            index},1081        {"content",          content},1082        {"tokens",           tokens},1083        {"stop",             false},1084        {"id_slot",          id_slot},1085        {"tokens_predicted", n_decoded},1086        {"tokens_evaluated", n_prompt_tokens},1087    };1088    // populate the timings object when needed (usually for the last response or with timings_per_token enabled)1089    if (timings.prompt_n > 0) {1090        res.push_back({"timings", timings.to_json()});1091    }1092    if (is_progress) {1093        res.push_back({"prompt_progress", progress.to_json()});1094    }1095    if (!prob_output.probs.empty()) {1096        res["completion_probabilities"] = completion_token_output::probs_vector_to_json({prob_output}, post_sampling_probs);1097    }1098    return res;1099}1100 1101json server_task_result_cmpl_partial::to_json_oaicompat() {1102    std::time_t t = std::time(0);1103    json logprobs = json(nullptr); // OAI default to null1104    if (prob_output.probs.size() > 0) {1105        logprobs = json{1106            {"content", completion_token_output::probs_vector_to_json({prob_output}, post_sampling_probs)},1107        };1108    }1109    json res = json {1110        {"choices",            json::array({1111            json{1112                {"text",          content},1113                {"index",         index},1114                {"logprobs",      logprobs},1115                {"finish_reason", nullptr},1116            }1117        })},1118        {"created",            t},1119        {"model",              oaicompat_model},1120        {"system_fingerprint", std::string(llama_build_info())},1121        {"object",             "text_completion"},1122        {"id",                 oaicompat_cmpl_id}1123    };1124 1125    // extra fields for debugging purposes1126    if (verbose) {1127        res["__verbose"] = to_json_non_oaicompat();1128    }1129    if (timings.prompt_n >= 0) {1130        res.push_back({"timings", timings.to_json()});1131    }1132    if (is_progress) {1133        res.push_back({"prompt_progress", progress.to_json()});1134    }1135 1136    return res;1137}1138 1139json server_task_result_cmpl_partial::to_json_oaicompat_chat() {1140    bool first = n_decoded == 1;1141    std::time_t t = std::time(0);1142    json choices;1143 1144    std::vector<json> deltas;1145    auto add_delta = [&](const json & delta) {1146        deltas.push_back({1147            {"choices", json::array({1148                json {1149                    {"finish_reason", nullptr},1150                    {"index", index},1151                    {"delta", delta},1152                },1153            })},1154            {"created", t},1155            {"id", oaicompat_cmpl_id},1156            {"model", oaicompat_model},1157            {"system_fingerprint", std::string(llama_build_info())},1158            {"object", "chat.completion.chunk"},1159        });1160    };1161    // We have to send an initial update to conform to openai behavior1162    if (first || is_progress) {1163        add_delta({1164            {"role", "assistant"},1165            {"content", nullptr},1166        });1167    }1168 1169    for (const auto & diff : oaicompat_msg_diffs) {1170        add_delta(server_chat_msg_diff_to_json_oaicompat(diff));1171    }1172 1173    if (!deltas.empty()) {1174        auto & last_json = deltas[deltas.size() - 1];1175        GGML_ASSERT(last_json.at("choices").size() >= 1);1176 1177        if (prob_output.probs.size() > 0) {1178            last_json.at("choices").at(0)["logprobs"] = json {1179                {"content", completion_token_output::probs_vector_to_json({prob_output}, post_sampling_probs)},1180            };1181        }1182 1183        if (timings.prompt_n >= 0) {1184            last_json.push_back({"timings", timings.to_json()});1185        }1186        if (is_progress) {1187            last_json.push_back({"prompt_progress", progress.to_json()});1188        }1189    }1190 1191    return deltas;1192}1193 1194json server_task_result_cmpl_partial::to_json_oaicompat_resp() {1195    std::vector<json> events;1196 1197    if (!oai_resp_created) {1198        events.push_back(json {1199            {"event", "response.created"},1200            {"data", json {

Showing the first 1,200 of 1855 lines. Download the file for the rest.

Brunobkr/llama.cpp_AlgMor24_github · Team Ai