Felipe97/llama-cpp-compiled
01.2k
1#include "parsers.h"2 3// Kimi K3 - XTML tagged format, built by open_tag/close_tag macros:4// open_tag(t, attrs) = <|open|>t k="v"...<|sep|> close_tag(t) = <|close|>t<|sep|>5// assistant := [think] [response] [tools] close_tag(message) <|end_of_msg|>6// the generation prompt already opens the think (or response) section, so the7// section opener is optional here - same as Kimi K2 Thinking8common_chat_params common_chat_params_init_kimi_k3(const common_chat_template & tmpl,9 const autoparser::generation_params & inputs) {10 common_chat_params data;11 12 data.prompt = common_chat_template_direct_apply_impl(tmpl, inputs);13 data.generation_prompt = common_chat_template_generation_prompt_impl(tmpl, inputs);14 data.format = COMMON_CHAT_FORMAT_PEG_NATIVE;15 data.supports_thinking = true;16 17 const std::string SEP = "<|sep|>";18 const std::string MSG_START = "<|open|>message role=\"assistant\"<|sep|>";19 const std::string THINK_START = "<|open|>think<|sep|>";20 const std::string THINK_END = "<|close|>think<|sep|>";21 const std::string RESP_START = "<|open|>response<|sep|>";22 const std::string RESP_END = "<|close|>response<|sep|>";23 const std::string TOOLS_START = "<|open|>tools<|sep|>";24 const std::string TOOLS_END = "<|close|>tools<|sep|>";25 const std::string CALL_START = "<|open|>call tool=\"";26 const std::string CALL_END = "<|close|>call<|sep|>";27 const std::string ARG_START = "<|open|>argument key=\"";28 const std::string ARG_END = "<|close|>argument<|sep|>";29 const std::string MSG_END = "<|close|>message<|sep|>";30 const std::string EOM_TOKEN = "<|end_of_msg|>";31 32 // only the markers are special tokens. tag names ("think", "response", ...) are33 // normal tokens and must not be preserved, or prose with those words is broken34 data.preserved_tokens = {35 "<|open|>",36 "<|close|>",37 "<|sep|>",38 "<|end_of_msg|>",39 };40 41 data.thinking_start_tag = THINK_START;42 data.thinking_end_tags = { THINK_END };43 44 // per-role message-start delimiters. user/assistant messages only have the role45 // attribute, so the full opener is used. system and tool messages have more46 // attributes, so those delimiters stop after the closing quote of the role47 data.message_delimiters = {48 { COMMON_CHAT_ROLE_ASSISTANT, "<|open|>message role=\"assistant\"<|sep|>" },49 { COMMON_CHAT_ROLE_USER, "<|open|>message role=\"user\"<|sep|>" },50 { COMMON_CHAT_ROLE_TOOL, "<|open|>message role=\"tool\"" },51 { COMMON_CHAT_ROLE_SYSTEM, "<|open|>message role=\"system\"" },52 };53 54 auto has_tools = inputs.tools.is_array() && !inputs.tools.empty();55 auto extract_reasoning = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;56 auto include_grammar = has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE;57 58 if (inputs.has_continuation()) {59 const auto & msg = inputs.continue_msg;60 61 data.generation_prompt = MSG_START + THINK_START + msg.reasoning_content;62 if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {63 data.generation_prompt += THINK_END + RESP_START + msg.render_content();64 }65 66 data.prompt += data.generation_prompt;67 }68 69 auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {70 auto end = p.end();71 72 auto start = p.optional(p.literal(MSG_START));73 74 // the think section is always consumed, even with reasoning extraction off:75 // the generation prompt ends with open_tag('think'), so it is always present.76 // reasoning stops at its own closer, or at the response opener if the model77 // skips the closer78 auto think_body = extract_reasoning ? p.reasoning(p.until_one_of({ THINK_END, RESP_START })) :79 p.content(p.until_one_of({ THINK_END, RESP_START }));80 81 auto reasoning = p.optional(p.optional(p.literal(THINK_START)) + think_body +82 p.optional(p.literal(THINK_END)));83 84 // content runs to the response closer, or to the next section if truncated85 auto response = p.optional(p.literal(RESP_START)) +86 p.content(p.until_one_of({ RESP_END, TOOLS_START, MSG_END })) +87 p.optional(p.literal(RESP_END));88 89 // the EOG token after the message closer reaches the parser as text,90 // so it must be consumed or the parse stays incomplete91 auto trailer = p.optional(p.literal(MSG_END)) + p.optional(p.literal(EOM_TOKEN));92 93 if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {94 return start + reasoning + response + trailer + end;95 }96 97 auto tool_choices = p.choice();98 foreach_function(inputs.tools, [&](const json & tool) {99 const auto & function = tool.at("function");100 std::string name = function.at("name");101 const json schema = common_chat_tool_parameters(function);102 103 // arguments come one tag per key, with the JSON type in a type="..."104 // attribute. the type is taken from the tool schema instead, as it tells105 // us if the value is JSON or a literal string106 auto args = p.eps();107 if (schema.contains("properties") && !schema.at("properties").empty()) {108 auto arg_choices = p.choice();109 for (const auto & prop : schema.at("properties").items()) {110 const std::string & key = prop.key();111 112 std::string type = "string";113 if (prop.value().is_object() && prop.value().contains("type") &&114 prop.value().at("type").is_string()) {115 type = prop.value().at("type").get<std::string>();116 }117 118 auto value = type == "string" ? p.tool_arg_string_value(p.until(ARG_END)) :119 p.tool_arg_value(p.until(ARG_END));120 121 // skip the trailing type="..." attribute: anything up to <|sep|>122 arg_choices |= p.rule("kimi-k3-arg-" + name + "-" + key,123 p.tool_arg(p.tool_arg_open(p.literal(ARG_START)) +124 p.tool_arg_name(p.literal(key)) + p.literal("\"") +125 p.until(SEP) + p.literal(SEP) + value +126 p.tool_arg_close(p.literal(ARG_END))));127 }128 args = p.zero_or_more(arg_choices);129 }130 131 // skip the trailing index="N" attribute the same way132 auto call = p.tool(p.tool_open(p.literal(CALL_START) + p.tool_name(p.literal(name)) + p.literal("\"") +133 p.until(SEP) + p.literal(SEP)) +134 p.tool_args(args) + p.tool_close(p.literal(CALL_END)));135 136 tool_choices |= p.rule("kimi-k3-tool-" + name, call);137 });138 139 // all calls go inside one tools section, then the message is closed. the140 // message closer is part of the trigger rule, or else the lazy grammar141 // rejects it once tool calls have started142 auto tools_section =143 p.trigger_rule("kimi-k3-tool-call", p.literal(TOOLS_START) + p.one_or_more(tool_choices) +144 p.literal(TOOLS_END) + p.optional(p.literal(MSG_END)) +145 p.optional(p.literal(EOM_TOKEN)));146 147 auto tools = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED ? tools_section :148 p.optional(tools_section);149 150 return start + reasoning + response + tools + trailer + end;151 });152 153 data.parser = parser.save();154 155 if (include_grammar) {156 data.grammar_lazy = inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_REQUIRED;157 data.grammar = build_grammar([&](const common_grammar_builder & builder) {158 parser.build_grammar(builder, data.grammar_lazy);159 });160 161 data.grammar_triggers = {162 { COMMON_GRAMMAR_TRIGGER_TYPE_WORD, TOOLS_START },163 };164 }165 166 return data;167}168 