Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
cohere2moe.cpp142 linesDownload Raw Back to parsers
1#include "parsers.h"2 3// Cohere2 MoE (a.k.a. "North Code") parser.4//5// The assistant turn is fully marker-wrapped:6//   <|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|>7//     <|START_THINKING|>{reasoning}<|END_THINKING|>8//     then EITHER content:    <|START_TEXT|>{content}<|END_TEXT|>9//          OR     tool calls: <|START_ACTION|>[10//                                 {"tool_call_id": "0", "tool_name": "f", "parameters": {...}}, ...11//                             ]<|END_ACTION|>12//   <|END_OF_TURN_TOKEN|>13//14// The generation prompt forces a leading <|START_THINKING|> (when reasoning is enabled, which is15// the template default), so the model's output continues from *inside* the thinking block. The16// parser literal therefore only covers the stable <|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|> prefix17// and the reasoning rule consumes the <|START_THINKING|> ... <|END_THINKING|> markers itself,18// regardless of whether they came from the generation prompt or the generated text.19common_chat_params common_chat_params_init_cohere2moe(const common_chat_template &          tmpl,20                                                              const autoparser::generation_params & inputs) {21    common_chat_params data;22 23    const std::string TURN_START    = "<|START_OF_TURN_TOKEN|>";24    const std::string TURN_END      = "<|END_OF_TURN_TOKEN|>";25    const std::string CHATBOT       = "<|CHATBOT_TOKEN|>";26    const std::string USER          = "<|USER_TOKEN|>";27    const std::string SYSTEM        = "<|SYSTEM_TOKEN|>";28    const std::string THINK_START   = "<|START_THINKING|>";29    const std::string THINK_END     = "<|END_THINKING|>";30    const std::string TEXT_START    = "<|START_TEXT|>";31    const std::string TEXT_END      = "<|END_TEXT|>";32    const std::string ACTION_START  = "<|START_ACTION|>";33    const std::string ACTION_END    = "<|END_ACTION|>";34    const std::string RESULT_START  = "<|START_TOOL_RESULT|>";35    const std::string RESULT_END    = "<|END_TOOL_RESULT|>";36 37    // Stable prefix of the generation prompt that precedes the (forced) <|START_THINKING|> marker.38    const std::string GEN_PREFIX = TURN_START + CHATBOT;39 40    data.prompt             = common_chat_template_direct_apply_impl(tmpl, inputs);41    data.generation_prompt  = common_chat_template_generation_prompt_impl(tmpl, inputs);42    data.format             = COMMON_CHAT_FORMAT_PEG_NATIVE;43    data.supports_thinking  = true;44    data.thinking_start_tag = THINK_START;45    data.thinking_end_tags  = {THINK_END};46    data.preserved_tokens   = {47        TURN_START, TURN_END, CHATBOT, USER, SYSTEM,48        THINK_START, THINK_END,49        TEXT_START, TEXT_END,50        ACTION_START, ACTION_END,51        RESULT_START, RESULT_END,52    };53 54    // Declare per-role message delimiters. Tool results are rendered with the55    // system token followed by <|START_TOOL_RESULT|>, so the "tool" delimiter must be listed before56    // the plain "system" one (it is a strict superset, and the role split tries delimiters in order).57    data.message_delimiters = {58        { COMMON_CHAT_ROLE_ASSISTANT, GEN_PREFIX },59        { COMMON_CHAT_ROLE_USER,      TURN_START + USER },60        { COMMON_CHAT_ROLE_TOOL,      TURN_START + SYSTEM + RESULT_START },61        { COMMON_CHAT_ROLE_SYSTEM,    TURN_START + SYSTEM },62    };63 64    auto has_tools           = inputs.tools.is_array() && !inputs.tools.empty();65    auto has_response_format = inputs.json_schema.is_object() && !inputs.json_schema.empty();66    auto extract_reasoning   = inputs.reasoning_format != COMMON_REASONING_FORMAT_NONE;67    auto include_grammar     = has_response_format || (has_tools && inputs.tool_choice != COMMON_CHAT_TOOL_CHOICE_NONE);68 69    if (inputs.has_continuation()) {70        const auto & msg = inputs.continue_msg;71 72        data.generation_prompt = GEN_PREFIX + THINK_START + msg.reasoning_content;73        if (inputs.continue_final_message == COMMON_CHAT_CONTINUATION_CONTENT) {74            data.generation_prompt += THINK_END + TEXT_START + msg.render_content();75        }76 77        data.prompt += data.generation_prompt;78    }79 80    auto parser = build_chat_peg_parser([&](common_chat_peg_builder & p) {81        auto generation_prompt = p.literal(GEN_PREFIX);82        auto end               = p.end();83 84        // The thinking block is always present (the generation prompt forces <|START_THINKING|>).85        // When extracting reasoning, capture its body; otherwise keep the whole block (markers86        // included) inline as content, matching reasoning_format=NONE conventions.87        common_peg_parser reasoning = p.eps();88        if (extract_reasoning) {89            reasoning = p.optional(p.literal(THINK_START) +90                                   p.reasoning(p.until_one_of({ THINK_END, TEXT_START, ACTION_START })) +91                                   p.optional(p.literal(THINK_END)));92        } else {93            reasoning = p.optional(p.content(p.literal(THINK_START) +94                                             p.until_one_of({ THINK_END, TEXT_START, ACTION_START }) +95                                             p.optional(p.literal(THINK_END))));96        }97 98        auto text_content = has_response_format99            ? p.literal(TEXT_START) +100                p.content(p.schema(p.json(), "response-format-schema", inputs.json_schema)) +101                p.optional(p.literal(TEXT_END))102            : p.literal(TEXT_START) + p.content(p.until(TEXT_END)) + p.optional(p.literal(TEXT_END));103 104        if (!has_tools || inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_NONE) {105            return generation_prompt + reasoning + text_content + p.optional(p.literal(TURN_END)) + end;106        }107 108        auto require_tools = inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_REQUIRED;109 110        // <|START_ACTION|>[ {"tool_call_id": "0", "tool_name": "f", "parameters": {...}}, ... ]<|END_ACTION|>111        auto tool_calls = p.standard_json_tools(ACTION_START, ACTION_END, inputs.tools, inputs.parallel_tool_calls,112                                                /* force_tool_calls = */ true,113                                                /* name_key         = */ "tool_name",114                                                /* args_key         = */ "parameters",115                                                /* array_wrapped    = */ true,116                                                /* function_is_key  = */ false,117                                                /* call_id_key      = */ "",118                                                /* gen_call_id_key  = */ "tool_call_id",119                                                /* parameters_order = */ { "tool_call_id", "tool_name", "parameters" });120 121        // Content and tool calls are mutually exclusive in this format.122        common_peg_parser body = require_tools ? tool_calls : p.choice({ tool_calls, text_content });123 124        return generation_prompt + reasoning + body + p.optional(p.literal(TURN_END)) + end;125    });126 127    data.parser = parser.save();128 129    if (include_grammar) {130        data.grammar_lazy = !has_response_format && inputs.tool_choice == COMMON_CHAT_TOOL_CHOICE_AUTO;131        data.grammar      = build_grammar([&](const common_grammar_builder & builder) {132            parser.build_grammar(builder, data.grammar_lazy);133        });134 135        data.grammar_triggers = {136            { COMMON_GRAMMAR_TRIGGER_TYPE_WORD, ACTION_START }137        };138    }139 140    return data;141}142