Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
kimi-k3.cpp168 linesDownload Raw Back to parsers
1#include "parsers.h"2 3// Kimi K3 - XTML tagged format, built by open_tag/close_tag macros:4//   open_tag(t, attrs) = <|open|>t k="v"...<|sep|>   close_tag(t) = <|close|>t<|sep|>5//   assistant := [think] [response] [tools] close_tag(message) <|end_of_msg|>6// the generation prompt already opens the think (or response) section, so the7// section opener is optional here - same as Kimi K2 Thinking8common_chat_params common_chat_params_init_kimi_k3(const common_chat_template &          tmpl,9                                                          const autoparser::generation_params & inputs) {10    common_chat_params data;11 12    data.prompt            = common_chat_template_direct_apply_impl(tmpl, inputs);13    data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, inputs);14    data.format            = COMMON_CHAT_FORMAT_PEG_NATIVE;15    data.supports_thinking = true;16 17    const std::string SEP         = "<|sep|>";18    const std::string MSG_START   = "<|open|>message role=\"assistant\"<|sep|>";19    const std::string THINK_START = "<|open|>think<|sep|>";20    const std::string THINK_END   = "<|close|>think<|sep|>";21    const std::string RESP_START  = "<|open|>response<|sep|>";22    const std::string RESP_END    = "<|close|>response<|sep|>";23    const std::string TOOLS_START = "<|open|>tools<|sep|>";24    const std::string TOOLS_END   = "<|close|>tools<|sep|>";25    const std::string CALL_START  = "<|open|>call tool=\"";26    const std::string CALL_END    = "<|close|>call<|sep|>";27    const std::string ARG_START   = "<|open|>argument key=\"";28    const std::string ARG_END     = "<|close|>argument<|sep|>";29    const std::string MSG_END     = "<|close|>message<|sep|>";30    const std::string EOM_TOKEN   = "<|end_of_msg|>";31 32    // only the markers are special tokens. tag names ("think", "response", ...) are33    // normal tokens and must not be preserved, or prose with those words is broken34    data.preserved_tokens = {35        "<|open|>",36        "<|close|>",37        "<|sep|>",38        "<|end_of_msg|>",39    };40 41    data.thinking_start_tag = THINK_START;42    data.thinking_end_tags  = { THINK_END };43 44    // per-role message-start delimiters. user/assistant messages only have the role45    // attribute, so the full opener is used. system and tool messages have more46    // attributes, so those delimiters stop after the closing quote of the role47    data.message_delimiters = {48        { COMMON_CHAT_ROLE_ASSISTANT, "<|open|>message role=\"assistant\"<|sep|>" },49        { COMMON_CHAT_ROLE_USER,      "<|open|>message role=\"user\"<|sep|>"      },50        { COMMON_CHAT_ROLE_TOOL,      "<|open|>message role=\"tool\""             },51        { COMMON_CHAT_ROLE_SYSTEM,    "<|open|>message role=\"system\""           },52    };53 54    auto has_tools         = inputs.tools.is_array() && !inputs.tools.empty();55    auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;56    auto include_grammar   = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;57 58    if (inputs.has_continuation()) {59        const auto & msg = inputs.continue_msg;60 61        data.generation_prompt = MSG_START + THINK_START + msg.reasoning_content;62        if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {63            data.generation_prompt += THINK_END + RESP_START + msg.render_content();64        }65 66        data.prompt += data.generation_prompt;67    }68 69    auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {70        auto end = p.end();71 72        auto start = p.optional(p.literal(MSG_START));73 74        // the think section is always consumed, even with reasoning extraction off:75        // the generation prompt ends with open_tag('think'), so it is always present.76        // reasoning stops at its own closer, or at the response opener if the model77        // skips the closer78        auto think_body = extract_reasoning ? p.reasoning(p.until_one_of({ THINK_END, RESP_START })) :79                                              p.content(p.until_one_of({ THINK_END, RESP_START }));80 81        auto reasoning = p.optional(p.optional(p.literal(THINK_START)) + think_body +82                                    p.optional(p.literal(THINK_END)));83 84        // content runs to the response closer, or to the next section if truncated85        auto response = p.optional(p.literal(RESP_START)) +86                        p.content(p.until_one_of({ RESP_END, TOOLS_START, MSG_END })) +87                        p.optional(p.literal(RESP_END));88 89        // the EOG token after the message closer reaches the parser as text,90        // so it must be consumed or the parse stays incomplete91        auto trailer = p.optional(p.literal(MSG_END)) + p.optional(p.literal(EOM_TOKEN));92 93        if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {94            return start + reasoning + response + trailer + end;95        }96 97        auto tool_choices = p.choice();98        foreach_function(inputs.tools, [&](const json & tool) {99            const auto & function = tool.at("function");100            std::string  name     = function.at("name");101            const json   schema   = common_chat_tool_parameters(function);102 103            // arguments come one tag per key, with the JSON type in a type="..."104            // attribute. the type is taken from the tool schema instead, as it tells105            // us if the value is JSON or a literal string106            auto args = p.eps();107            if (schema.contains("properties") && !schema.at("properties").empty()) {108                auto arg_choices = p.choice();109                for (const auto & prop : schema.at("properties").items()) {110                    const std::string & key = prop.key();111 112                    std::string type = "string";113                    if (prop.value().is_object() && prop.value().contains("type") &&114                        prop.value().at("type").is_string()) {115                        type = prop.value().at("type").get<std::string>();116                    }117 118                    auto value = type == "string" ? p.tool_arg_string_value(p.until(ARG_END)) :119                                                    p.tool_arg_value(p.until(ARG_END));120 121                    // skip the trailing type="..." attribute: anything up to <|sep|>122                    arg_choices |= p.rule("kimi-k3-arg-" + name + "-" + key,123                                          p.tool_arg(p.tool_arg_open(p.literal(ARG_START)) +124                                                     p.tool_arg_name(p.literal(key)) + p.literal("\"") +125                                                     p.until(SEP) + p.literal(SEP) + value +126                                                     p.tool_arg_close(p.literal(ARG_END))));127                }128                args = p.zero_or_more(arg_choices);129            }130 131            // skip the trailing index="N" attribute the same way132            auto call = p.tool(p.tool_open(p.literal(CALL_START) + p.tool_name(p.literal(name)) + p.literal("\"") +133                                           p.until(SEP) + p.literal(SEP)) +134                               p.tool_args(args) + p.tool_close(p.literal(CALL_END)));135 136            tool_choices |= p.rule("kimi-k3-tool-" + name, call);137        });138 139        // all calls go inside one tools section, then the message is closed. the140        // message closer is part of the trigger rule, or else the lazy grammar141        // rejects it once tool calls have started142        auto tools_section =143            p.trigger_rule("kimi-k3-tool-call", p.literal(TOOLS_START) + p.one_or_more(tool_choices) +144                                                    p.literal(TOOLS_END) + p.optional(p.literal(MSG_END)) +145                                                    p.optional(p.literal(EOM_TOKEN)));146 147        auto tools = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED ? tools_section :148                                                                              p.optional(tools_section);149 150        return start + reasoning + response + tools + trailer + end;151    });152 153    data.parser = parser.save();154 155    if (include_grammar) {156        data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;157        data.grammar      = build_grammar([&](const common_grammar_builder & builder) {158            parser.build_grammar(builder, data.grammar_lazy);159        });160 161        data.grammar_triggers = {162            { COMMON_GRAMMAR_TRIGGER_TYPE_WORD, TOOLS_START },163        };164    }165 166    return data;167}168