KBaba7/llama.cpp
0
1#include "llama-chat.h"2 3#include "llama.h"4 5#include <map>6#include <sstream>7 8#if __cplusplus >= 202000L9 #define LU8(x) (const char*)(u8##x)10#else11 #define LU8(x) u8##x12#endif13 14// trim whitespace from the beginning and end of a string15static std::string trim(const std::string & str) {16 size_t start = 0;17 size_t end = str.size();18 while (start < end && isspace(str[start])) {19 start += 1;20 }21 while (end > start && isspace(str[end - 1])) {22 end -= 1;23 }24 return str.substr(start, end - start);25}26 27static const std::map<std::string, llm_chat_template> LLM_CHAT_TEMPLATES = {28 { "chatml", LLM_CHAT_TEMPLATE_CHATML },29 { "llama2", LLM_CHAT_TEMPLATE_LLAMA_2 },30 { "llama2-sys", LLM_CHAT_TEMPLATE_LLAMA_2_SYS },31 { "llama2-sys-bos", LLM_CHAT_TEMPLATE_LLAMA_2_SYS_BOS },32 { "llama2-sys-strip", LLM_CHAT_TEMPLATE_LLAMA_2_SYS_STRIP },33 { "mistral-v1", LLM_CHAT_TEMPLATE_MISTRAL_V1 },34 { "mistral-v3", LLM_CHAT_TEMPLATE_MISTRAL_V3 },35 { "mistral-v3-tekken", LLM_CHAT_TEMPLATE_MISTRAL_V3_TEKKEN },36 { "mistral-v7", LLM_CHAT_TEMPLATE_MISTRAL_V7 },37 { "phi3", LLM_CHAT_TEMPLATE_PHI_3 },38 { "phi4", LLM_CHAT_TEMPLATE_PHI_4 },39 { "falcon3", LLM_CHAT_TEMPLATE_FALCON_3 },40 { "zephyr", LLM_CHAT_TEMPLATE_ZEPHYR },41 { "monarch", LLM_CHAT_TEMPLATE_MONARCH },42 { "gemma", LLM_CHAT_TEMPLATE_GEMMA },43 { "orion", LLM_CHAT_TEMPLATE_ORION },44 { "openchat", LLM_CHAT_TEMPLATE_OPENCHAT },45 { "vicuna", LLM_CHAT_TEMPLATE_VICUNA },46 { "vicuna-orca", LLM_CHAT_TEMPLATE_VICUNA_ORCA },47 { "deepseek", LLM_CHAT_TEMPLATE_DEEPSEEK },48 { "deepseek2", LLM_CHAT_TEMPLATE_DEEPSEEK_2 },49 { "deepseek3", LLM_CHAT_TEMPLATE_DEEPSEEK_3 },50 { "command-r", LLM_CHAT_TEMPLATE_COMMAND_R },51 { "llama3", LLM_CHAT_TEMPLATE_LLAMA_3 },52 { "chatglm3", LLM_CHAT_TEMPLATE_CHATGML_3 },53 { "chatglm4", LLM_CHAT_TEMPLATE_CHATGML_4 },54 { "glmedge", LLM_CHAT_TEMPLATE_GLMEDGE },55 { "minicpm", LLM_CHAT_TEMPLATE_MINICPM },56 { "exaone3", LLM_CHAT_TEMPLATE_EXAONE_3 },57 { "rwkv-world", LLM_CHAT_TEMPLATE_RWKV_WORLD },58 { "granite", LLM_CHAT_TEMPLATE_GRANITE },59 { "gigachat", LLM_CHAT_TEMPLATE_GIGACHAT },60 { "megrez", LLM_CHAT_TEMPLATE_MEGREZ },61};62 63llm_chat_template llm_chat_template_from_str(const std::string & name) {64 return LLM_CHAT_TEMPLATES.at(name);65}66 67llm_chat_template llm_chat_detect_template(const std::string & tmpl) {68 try {69 return llm_chat_template_from_str(tmpl);70 } catch (const std::out_of_range &) {71 // ignore72 }73 74 auto tmpl_contains = [&tmpl](const char * haystack) -> bool {75 return tmpl.find(haystack) != std::string::npos;76 };77 if (tmpl_contains("<|im_start|>")) {78 return tmpl_contains("<|im_sep|>")79 ? LLM_CHAT_TEMPLATE_PHI_480 : LLM_CHAT_TEMPLATE_CHATML;81 } else if (tmpl.find("mistral") == 0 || tmpl_contains("[INST]")) {82 if (tmpl_contains("[SYSTEM_PROMPT]")) {83 return LLM_CHAT_TEMPLATE_MISTRAL_V7;84 } else if (85 // catches official 'v1' template86 tmpl_contains("' [INST] ' + system_message")87 // catches official 'v3' and 'v3-tekken' templates88 || tmpl_contains("[AVAILABLE_TOOLS]")89 ) {90 // Official mistral 'v1', 'v3' and 'v3-tekken' templates91 // See: https://github.com/mistralai/cookbook/blob/main/concept-deep-dive/tokenization/chat_templates.md92 // See: https://github.com/mistralai/cookbook/blob/main/concept-deep-dive/tokenization/templates.md93 if (tmpl_contains(" [INST]")) {94 return LLM_CHAT_TEMPLATE_MISTRAL_V1;95 } else if (tmpl_contains("\"[INST]\"")) {96 return LLM_CHAT_TEMPLATE_MISTRAL_V3_TEKKEN;97 }98 return LLM_CHAT_TEMPLATE_MISTRAL_V3;99 } else {100 // llama2 template and its variants101 // [variant] support system message102 // See: https://huggingface.co/blog/llama2#how-to-prompt-llama-2103 bool support_system_message = tmpl_contains("<<SYS>>");104 bool add_bos_inside_history = tmpl_contains("bos_token + '[INST]");105 bool strip_message = tmpl_contains("content.strip()");106 if (strip_message) {107 return LLM_CHAT_TEMPLATE_LLAMA_2_SYS_STRIP;108 } else if (add_bos_inside_history) {109 return LLM_CHAT_TEMPLATE_LLAMA_2_SYS_BOS;110 } else if (support_system_message) {111 return LLM_CHAT_TEMPLATE_LLAMA_2_SYS;112 } else {113 return LLM_CHAT_TEMPLATE_LLAMA_2;114 }115 }116 } else if (tmpl_contains("<|assistant|>") && tmpl_contains("<|end|>")) {117 return LLM_CHAT_TEMPLATE_PHI_3;118 } else if (tmpl_contains("<|assistant|>") && tmpl_contains("<|user|>")) {119 return tmpl_contains("</s>") ? LLM_CHAT_TEMPLATE_FALCON_3 : LLM_CHAT_TEMPLATE_GLMEDGE;120 } else if (tmpl_contains("<|user|>") && tmpl_contains("<|endoftext|>")) {121 return LLM_CHAT_TEMPLATE_ZEPHYR;122 } else if (tmpl_contains("bos_token + message['role']")) {123 return LLM_CHAT_TEMPLATE_MONARCH;124 } else if (tmpl_contains("<start_of_turn>")) {125 return LLM_CHAT_TEMPLATE_GEMMA;126 } else if (tmpl_contains("'\\n\\nAssistant: ' + eos_token")) {127 // OrionStarAI/Orion-14B-Chat128 return LLM_CHAT_TEMPLATE_ORION;129 } else if (tmpl_contains("GPT4 Correct ")) {130 // openchat/openchat-3.5-0106131 return LLM_CHAT_TEMPLATE_OPENCHAT;132 } else if (tmpl_contains("USER: ") && tmpl_contains("ASSISTANT: ")) {133 // eachadea/vicuna-13b-1.1 (and Orca variant)134 if (tmpl_contains("SYSTEM: ")) {135 return LLM_CHAT_TEMPLATE_VICUNA_ORCA;136 }137 return LLM_CHAT_TEMPLATE_VICUNA;138 } else if (tmpl_contains("### Instruction:") && tmpl_contains("<|EOT|>")) {139 // deepseek-ai/deepseek-coder-33b-instruct140 return LLM_CHAT_TEMPLATE_DEEPSEEK;141 } else if (tmpl_contains("<|START_OF_TURN_TOKEN|>") && tmpl_contains("<|USER_TOKEN|>")) {142 // CohereForAI/c4ai-command-r-plus143 return LLM_CHAT_TEMPLATE_COMMAND_R;144 } else if (tmpl_contains("<|start_header_id|>") && tmpl_contains("<|end_header_id|>")) {145 return LLM_CHAT_TEMPLATE_LLAMA_3;146 } else if (tmpl_contains("[gMASK]sop")) {147 // chatglm3-6b148 return LLM_CHAT_TEMPLATE_CHATGML_3;149 } else if (tmpl_contains("[gMASK]<sop>")) {150 return LLM_CHAT_TEMPLATE_CHATGML_4;151 } else if (tmpl_contains(LU8("<用户>"))) {152 // MiniCPM-3B-OpenHermes-2.5-v2-GGUF153 return LLM_CHAT_TEMPLATE_MINICPM;154 } else if (tmpl_contains("'Assistant: ' + message['content'] + eos_token")) {155 return LLM_CHAT_TEMPLATE_DEEPSEEK_2;156 } else if (tmpl_contains(LU8("<|Assistant|>")) && tmpl_contains(LU8("<|User|>")) && tmpl_contains(LU8("<|end▁of▁sentence|>"))) {157 return LLM_CHAT_TEMPLATE_DEEPSEEK_3;158 } else if (tmpl_contains("[|system|]") && tmpl_contains("[|assistant|]") && tmpl_contains("[|endofturn|]")) {159 // ref: https://huggingface.co/LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct/discussions/8#66bae61b1893d14ee8ed85bb160 // EXAONE-3.0-7.8B-Instruct161 return LLM_CHAT_TEMPLATE_EXAONE_3;162 } else if (tmpl_contains("rwkv-world")) {163 return LLM_CHAT_TEMPLATE_RWKV_WORLD;164 } else if (tmpl_contains("<|start_of_role|>")) {165 return LLM_CHAT_TEMPLATE_GRANITE;166 } else if (tmpl_contains("message['role'] + additional_special_tokens[0] + message['content'] + additional_special_tokens[1]")) {167 return LLM_CHAT_TEMPLATE_GIGACHAT;168 } else if (tmpl_contains("<|role_start|>")) {169 return LLM_CHAT_TEMPLATE_MEGREZ;170 }171 return LLM_CHAT_TEMPLATE_UNKNOWN;172}173 174// Simple version of "llama_apply_chat_template" that only works with strings175// This function uses heuristic checks to determine commonly used template. It is not a jinja parser.176int32_t llm_chat_apply_template(177 llm_chat_template tmpl,178 const std::vector<const llama_chat_message *> & chat,179 std::string & dest, bool add_ass) {180 // Taken from the research: https://github.com/ggerganov/llama.cpp/issues/5527181 std::stringstream ss;182 if (tmpl == LLM_CHAT_TEMPLATE_CHATML) {183 // chatml template184 for (auto message : chat) {185 ss << "<|im_start|>" << message->role << "\n" << message->content << "<|im_end|>\n";186 }187 if (add_ass) {188 ss << "<|im_start|>assistant\n";189 }190 } else if (tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V7) {191 // Official mistral 'v7' template192 // See: https://huggingface.co/mistralai/Mistral-Large-Instruct-2411#basic-instruct-template-v7193 for (auto message : chat) {194 std::string role(message->role);195 std::string content(message->content);196 if (role == "system") {197 ss << "[SYSTEM_PROMPT] " << content << "[/SYSTEM_PROMPT]";198 } else if (role == "user") {199 ss << "[INST] " << content << "[/INST]";200 }201 else {202 ss << " " << content << "</s>";203 }204 }205 } else if (tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V1206 || tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V3207 || tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V3_TEKKEN) {208 // See: https://github.com/mistralai/cookbook/blob/main/concept-deep-dive/tokenization/chat_templates.md209 // See: https://github.com/mistralai/cookbook/blob/main/concept-deep-dive/tokenization/templates.md210 std::string leading_space = tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V1 ? " " : "";211 std::string trailing_space = tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V3_TEKKEN ? "" : " ";212 bool trim_assistant_message = tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V3;213 bool is_inside_turn = false;214 for (auto message : chat) {215 if (!is_inside_turn) {216 ss << leading_space << "[INST]" << trailing_space;217 is_inside_turn = true;218 }219 std::string role(message->role);220 std::string content(message->content);221 if (role == "system") {222 ss << content << "\n\n";223 } else if (role == "user") {224 ss << content << leading_space << "[/INST]";225 } else {226 ss << trailing_space << (trim_assistant_message ? trim(content) : content) << "</s>";227 is_inside_turn = false;228 }229 }230 } else if (231 tmpl == LLM_CHAT_TEMPLATE_LLAMA_2232 || tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS233 || tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS_BOS234 || tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS_STRIP) {235 // llama2 template and its variants236 // [variant] support system message237 // See: https://huggingface.co/blog/llama2#how-to-prompt-llama-2238 bool support_system_message = tmpl != LLM_CHAT_TEMPLATE_LLAMA_2;239 // [variant] add BOS inside history240 bool add_bos_inside_history = tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS_BOS;241 // [variant] trim spaces from the input message242 bool strip_message = tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS_STRIP;243 // construct the prompt244 bool is_inside_turn = true; // skip BOS at the beginning245 ss << "[INST] ";246 for (auto message : chat) {247 std::string content = strip_message ? trim(message->content) : message->content;248 std::string role(message->role);249 if (!is_inside_turn) {250 is_inside_turn = true;251 ss << (add_bos_inside_history ? "<s>[INST] " : "[INST] ");252 }253 if (role == "system") {254 if (support_system_message) {255 ss << "<<SYS>>\n" << content << "\n<</SYS>>\n\n";256 } else {257 // if the model does not support system message, we still include it in the first message, but without <<SYS>>258 ss << content << "\n";259 }260 } else if (role == "user") {261 ss << content << " [/INST]";262 } else {263 ss << content << "</s>";264 is_inside_turn = false;265 }266 }267 } else if (tmpl == LLM_CHAT_TEMPLATE_PHI_3) {268 // Phi 3269 for (auto message : chat) {270 std::string role(message->role);271 ss << "<|" << role << "|>\n" << message->content << "<|end|>\n";272 }273 if (add_ass) {274 ss << "<|assistant|>\n";275 }276 } else if (tmpl == LLM_CHAT_TEMPLATE_PHI_4) {277 // chatml template278 for (auto message : chat) {279 ss << "<|im_start|>" << message->role << "<|im_sep|>" << message->content << "<|im_end|>";280 }281 if (add_ass) {282 ss << "<|im_start|>assistant<|im_sep|>";283 }284 } else if (tmpl == LLM_CHAT_TEMPLATE_FALCON_3) {285 // Falcon 3286 for (auto message : chat) {287 std::string role(message->role);288 ss << "<|" << role << "|>\n" << message->content << "\n";289 }290 if (add_ass) {291 ss << "<|assistant|>\n";292 }293 } else if (tmpl == LLM_CHAT_TEMPLATE_ZEPHYR) {294 // zephyr template295 for (auto message : chat) {296 ss << "<|" << message->role << "|>" << "\n" << message->content << "<|endoftext|>\n";297 }298 if (add_ass) {299 ss << "<|assistant|>\n";300 }301 } else if (tmpl == LLM_CHAT_TEMPLATE_MONARCH) {302 // mlabonne/AlphaMonarch-7B template (the <s> is included inside history)303 for (auto message : chat) {304 std::string bos = (message == chat.front()) ? "" : "<s>"; // skip BOS for first message305 ss << bos << message->role << "\n" << message->content << "</s>\n";306 }307 if (add_ass) {308 ss << "<s>assistant\n";309 }310 } else if (tmpl == LLM_CHAT_TEMPLATE_GEMMA) {311 // google/gemma-7b-it312 std::string system_prompt = "";313 for (auto message : chat) {314 std::string role(message->role);315 if (role == "system") {316 // there is no system message for gemma, but we will merge it with user prompt, so nothing is broken317 system_prompt = trim(message->content);318 continue;319 }320 // in gemma, "assistant" is "model"321 role = role == "assistant" ? "model" : message->role;322 ss << "<start_of_turn>" << role << "\n";323 if (!system_prompt.empty() && role != "model") {324 ss << system_prompt << "\n\n";325 system_prompt = "";326 }327 ss << trim(message->content) << "<end_of_turn>\n";328 }329 if (add_ass) {330 ss << "<start_of_turn>model\n";331 }332 } else if (tmpl == LLM_CHAT_TEMPLATE_ORION) {333 // OrionStarAI/Orion-14B-Chat334 std::string system_prompt = "";335 for (auto message : chat) {336 std::string role(message->role);337 if (role == "system") {338 // there is no system message support, we will merge it with user prompt339 system_prompt = message->content;340 continue;341 } else if (role == "user") {342 ss << "Human: ";343 if (!system_prompt.empty()) {344 ss << system_prompt << "\n\n";345 system_prompt = "";346 }347 ss << message->content << "\n\nAssistant: </s>";348 } else {349 ss << message->content << "</s>";350 }351 }352 } else if (tmpl == LLM_CHAT_TEMPLATE_OPENCHAT) {353 // openchat/openchat-3.5-0106,354 for (auto message : chat) {355 std::string role(message->role);356 if (role == "system") {357 ss << message->content << "<|end_of_turn|>";358 } else {359 role[0] = toupper(role[0]);360 ss << "GPT4 Correct " << role << ": " << message->content << "<|end_of_turn|>";361 }362 }363 if (add_ass) {364 ss << "GPT4 Correct Assistant:";365 }366 } else if (tmpl == LLM_CHAT_TEMPLATE_VICUNA || tmpl == LLM_CHAT_TEMPLATE_VICUNA_ORCA) {367 // eachadea/vicuna-13b-1.1 (and Orca variant)368 for (auto message : chat) {369 std::string role(message->role);370 if (role == "system") {371 // Orca-Vicuna variant uses a system prefix372 if (tmpl == LLM_CHAT_TEMPLATE_VICUNA_ORCA) {373 ss << "SYSTEM: " << message->content << "\n";374 } else {375 ss << message->content << "\n\n";376 }377 } else if (role == "user") {378 ss << "USER: " << message->content << "\n";379 } else if (role == "assistant") {380 ss << "ASSISTANT: " << message->content << "</s>\n";381 }382 }383 if (add_ass) {384 ss << "ASSISTANT:";385 }386 } else if (tmpl == LLM_CHAT_TEMPLATE_DEEPSEEK) {387 // deepseek-ai/deepseek-coder-33b-instruct388 for (auto message : chat) {389 std::string role(message->role);390 if (role == "system") {391 ss << message->content;392 } else if (role == "user") {393 ss << "### Instruction:\n" << message->content << "\n";394 } else if (role == "assistant") {395 ss << "### Response:\n" << message->content << "\n<|EOT|>\n";396 }397 }398 if (add_ass) {399 ss << "### Response:\n";400 }401 } else if (tmpl == LLM_CHAT_TEMPLATE_COMMAND_R) {402 // CohereForAI/c4ai-command-r-plus403 for (auto message : chat) {404 std::string role(message->role);405 if (role == "system") {406 ss << "<|START_OF_TURN_TOKEN|><|SYSTEM_TOKEN|>" << trim(message->content) << "<|END_OF_TURN_TOKEN|>";407 } else if (role == "user") {408 ss << "<|START_OF_TURN_TOKEN|><|USER_TOKEN|>" << trim(message->content) << "<|END_OF_TURN_TOKEN|>";409 } else if (role == "assistant") {410 ss << "<|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|>" << trim(message->content) << "<|END_OF_TURN_TOKEN|>";411 }412 }413 if (add_ass) {414 ss << "<|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|>";415 }416 } else if (tmpl == LLM_CHAT_TEMPLATE_LLAMA_3) {417 // Llama 3418 for (auto message : chat) {419 std::string role(message->role);420 ss << "<|start_header_id|>" << role << "<|end_header_id|>\n\n" << trim(message->content) << "<|eot_id|>";421 }422 if (add_ass) {423 ss << "<|start_header_id|>assistant<|end_header_id|>\n\n";424 }425 } else if (tmpl == LLM_CHAT_TEMPLATE_CHATGML_3) {426 // chatglm3-6b427 ss << "[gMASK]" << "sop";428 for (auto message : chat) {429 std::string role(message->role);430 ss << "<|" << role << "|>" << "\n " << message->content;431 }432 if (add_ass) {433 ss << "<|assistant|>";434 }435 } else if (tmpl == LLM_CHAT_TEMPLATE_CHATGML_4) {436 ss << "[gMASK]" << "<sop>";437 for (auto message : chat) {438 std::string role(message->role);439 ss << "<|" << role << "|>" << "\n" << message->content;440 }441 if (add_ass) {442 ss << "<|assistant|>";443 }444 } else if (tmpl == LLM_CHAT_TEMPLATE_GLMEDGE) {445 for (auto message : chat) {446 std::string role(message->role);447 ss << "<|" << role << "|>" << "\n" << message->content;448 }449 if (add_ass) {450 ss << "<|assistant|>";451 }452 } else if (tmpl == LLM_CHAT_TEMPLATE_MINICPM) {453 // MiniCPM-3B-OpenHermes-2.5-v2-GGUF454 for (auto message : chat) {455 std::string role(message->role);456 if (role == "user") {457 ss << LU8("<用户>");458 ss << trim(message->content);459 ss << "<AI>";460 } else {461 ss << trim(message->content);462 }463 }464 } else if (tmpl == LLM_CHAT_TEMPLATE_DEEPSEEK_2) {465 // DeepSeek-V2466 for (auto message : chat) {467 std::string role(message->role);468 if (role == "system") {469 ss << message->content << "\n\n";470 } else if (role == "user") {471 ss << "User: " << message->content << "\n\n";472 } else if (role == "assistant") {473 ss << "Assistant: " << message->content << LU8("<|end▁of▁sentence|>");474 }475 }476 if (add_ass) {477 ss << "Assistant:";478 }479 } else if (tmpl == LLM_CHAT_TEMPLATE_DEEPSEEK_3) {480 // DeepSeek-V3481 for (auto message : chat) {482 std::string role(message->role);483 if (role == "system") {484 ss << message->content << "\n\n";485 } else if (role == "user") {486 ss << LU8("<|User|>") << message->content;487 } else if (role == "assistant") {488 ss << LU8("<|Assistant|>") << message->content << LU8("<|end▁of▁sentence|>");489 }490 }491 if (add_ass) {492 ss << LU8("<|Assistant|>");493 }494 } else if (tmpl == LLM_CHAT_TEMPLATE_EXAONE_3) {495 // ref: https://huggingface.co/LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct/discussions/8#66bae61b1893d14ee8ed85bb496 // EXAONE-3.0-7.8B-Instruct497 for (auto message : chat) {498 std::string role(message->role);499 if (role == "system") {500 ss << "[|system|]" << trim(message->content) << "[|endofturn|]\n";501 } else if (role == "user") {502 ss << "[|user|]" << trim(message->content) << "\n";503 } else if (role == "assistant") {504 ss << "[|assistant|]" << trim(message->content) << "[|endofturn|]\n";505 }506 }507 if (add_ass) {508 ss << "[|assistant|]";509 }510 } else if (tmpl == LLM_CHAT_TEMPLATE_RWKV_WORLD) {511 // this template requires the model to have "\n\n" as EOT token512 for (auto message : chat) {513 std::string role(message->role);514 if (role == "user") {515 ss << "User: " << message->content << "\n\nAssistant:";516 } else {517 ss << message->content << "\n\n";518 }519 }520 } else if (tmpl == LLM_CHAT_TEMPLATE_GRANITE) {521 // IBM Granite template522 for (const auto & message : chat) {523 std::string role(message->role);524 ss << "<|start_of_role|>" << role << "<|end_of_role|>";525 if (role == "assistant_tool_call") {526 ss << "<|tool_call|>";527 }528 ss << message->content << "<|end_of_text|>\n";529 }530 if (add_ass) {531 ss << "<|start_of_role|>assistant<|end_of_role|>\n";532 }533 } else if (tmpl == LLM_CHAT_TEMPLATE_GIGACHAT) {534 // GigaChat template535 bool has_system = !chat.empty() && std::string(chat[0]->role) == "system";536 537 // Handle system message if present538 if (has_system) {539 ss << "<s>" << chat[0]->content << "<|message_sep|>";540 } else {541 ss << "<s>";542 }543 544 // Process remaining messages545 for (size_t i = has_system ? 1 : 0; i < chat.size(); i++) {546 std::string role(chat[i]->role);547 if (role == "user") {548 ss << "user<|role_sep|>" << chat[i]->content << "<|message_sep|>"549 << "available functions<|role_sep|>[]<|message_sep|>";550 } else if (role == "assistant") {551 ss << "assistant<|role_sep|>" << chat[i]->content << "<|message_sep|>";552 }553 }554 555 // Add generation prompt if needed556 if (add_ass) {557 ss << "assistant<|role_sep|>";558 }559 } else if (tmpl == LLM_CHAT_TEMPLATE_MEGREZ) {560 // Megrez template561 for (auto message : chat) {562 std::string role(message->role);563 ss << "<|role_start|>" << role << "<|role_end|>" << message->content << "<|turn_end|>";564 }565 566 if (add_ass) {567 ss << "<|role_start|>assistant<|role_end|>";568 }569 } else {570 // template not supported571 return -1;572 }573 dest = ss.str();574 return dest.size();575}576 577// public interface578 579int32_t llama_chat_builtin_templates(const char ** output, size_t len) {580 auto it = LLM_CHAT_TEMPLATES.begin();581 for (size_t i = 0; i < std::min(len, LLM_CHAT_TEMPLATES.size()); i++) {582 output[i] = it->first.c_str();583 std::advance(it, 1);584 }585 return (int32_t) LLM_CHAT_TEMPLATES.size();586}587 588 