Team Ai
Datasetpublic

echodict/llama.cpp

version https://git-lfs.github.com/spec/v1 oid sha256:cfc44b7ba25614df70e6b65e3341cae0310163bd32fd31a6b928a542df433faf size 30786

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes579downloads
llama-chat.cpp940 linesDownload Raw Back to src
1#include "llama-chat.h"2 3#include "llama.h"4 5#include <map>6#include <sstream>7#include <algorithm>8 9#if __cplusplus >= 202000L10    #define LU8(x) (const char*)(u8##x)11#else12    #define LU8(x) u8##x13#endif14 15// trim whitespace from the beginning and end of a string16static std::string trim(const std::string & str) {17    size_t start = 0;18    size_t end = str.size();19    while (start < end && isspace(static_cast<unsigned char>(str[start]))) {20        start += 1;21    }22    while (end > start && isspace(static_cast<unsigned char>(str[end - 1]))) {23        end -= 1;24    }25    return str.substr(start, end - start);26}27 28static const std::map<std::string, llm_chat_template> LLM_CHAT_TEMPLATES = {29    { "chatml",            LLM_CHAT_TEMPLATE_CHATML            },30    { "llama2",            LLM_CHAT_TEMPLATE_LLAMA_2           },31    { "llama2-sys",        LLM_CHAT_TEMPLATE_LLAMA_2_SYS       },32    { "llama2-sys-bos",    LLM_CHAT_TEMPLATE_LLAMA_2_SYS_BOS   },33    { "llama2-sys-strip",  LLM_CHAT_TEMPLATE_LLAMA_2_SYS_STRIP },34    { "mistral-v1",        LLM_CHAT_TEMPLATE_MISTRAL_V1        },35    { "mistral-v3",        LLM_CHAT_TEMPLATE_MISTRAL_V3        },36    { "mistral-v3-tekken", LLM_CHAT_TEMPLATE_MISTRAL_V3_TEKKEN },37    { "mistral-v7",        LLM_CHAT_TEMPLATE_MISTRAL_V7        },38    { "mistral-v7-tekken", LLM_CHAT_TEMPLATE_MISTRAL_V7_TEKKEN },39    { "phi3",              LLM_CHAT_TEMPLATE_PHI_3             },40    { "phi4",              LLM_CHAT_TEMPLATE_PHI_4             },41    { "falcon3",           LLM_CHAT_TEMPLATE_FALCON_3          },42    { "zephyr",            LLM_CHAT_TEMPLATE_ZEPHYR            },43    { "monarch",           LLM_CHAT_TEMPLATE_MONARCH           },44    { "gemma",             LLM_CHAT_TEMPLATE_GEMMA             },45    { "orion",             LLM_CHAT_TEMPLATE_ORION             },46    { "openchat",          LLM_CHAT_TEMPLATE_OPENCHAT          },47    { "vicuna",            LLM_CHAT_TEMPLATE_VICUNA            },48    { "vicuna-orca",       LLM_CHAT_TEMPLATE_VICUNA_ORCA       },49    { "deepseek",          LLM_CHAT_TEMPLATE_DEEPSEEK          },50    { "deepseek2",         LLM_CHAT_TEMPLATE_DEEPSEEK_2        },51    { "deepseek3",         LLM_CHAT_TEMPLATE_DEEPSEEK_3        },52    { "deepseek-ocr",      LLM_CHAT_TEMPLATE_DEEPSEEK_OCR      },53    { "command-r",         LLM_CHAT_TEMPLATE_COMMAND_R         },54    { "llama3",            LLM_CHAT_TEMPLATE_LLAMA_3           },55    { "chatglm3",          LLM_CHAT_TEMPLATE_CHATGLM_3         },56    { "chatglm4",          LLM_CHAT_TEMPLATE_CHATGLM_4         },57    { "glmedge",           LLM_CHAT_TEMPLATE_GLMEDGE           },58    { "minicpm",           LLM_CHAT_TEMPLATE_MINICPM           },59    { "exaone3",           LLM_CHAT_TEMPLATE_EXAONE_3          },60    { "exaone4",           LLM_CHAT_TEMPLATE_EXAONE_4          },61    { "exaone-moe",        LLM_CHAT_TEMPLATE_EXAONE_MOE        },62    { "rwkv-world",        LLM_CHAT_TEMPLATE_RWKV_WORLD        },63    { "granite",           LLM_CHAT_TEMPLATE_GRANITE_3_X       },64    { "granite-4.0",       LLM_CHAT_TEMPLATE_GRANITE_4_0       },65    { "gigachat",          LLM_CHAT_TEMPLATE_GIGACHAT          },66    { "megrez",            LLM_CHAT_TEMPLATE_MEGREZ            },67    { "yandex",            LLM_CHAT_TEMPLATE_YANDEX            },68    { "bailing",           LLM_CHAT_TEMPLATE_BAILING           },69    { "bailing-think",     LLM_CHAT_TEMPLATE_BAILING_THINK     },70    { "bailing2",          LLM_CHAT_TEMPLATE_BAILING2          },71    { "llama4",            LLM_CHAT_TEMPLATE_LLAMA4            },72    { "smolvlm",           LLM_CHAT_TEMPLATE_SMOLVLM           },73    { "hunyuan-moe",       LLM_CHAT_TEMPLATE_HUNYUAN_MOE       },74    { "gpt-oss",           LLM_CHAT_TEMPLATE_OPENAI_MOE        },75    { "hunyuan-dense",     LLM_CHAT_TEMPLATE_HUNYUAN_DENSE     },76    { "hunyuan-ocr",       LLM_CHAT_TEMPLATE_HUNYUAN_OCR       },77    { "kimi-k2",           LLM_CHAT_TEMPLATE_KIMI_K2           },78    { "seed_oss",          LLM_CHAT_TEMPLATE_SEED_OSS          },79    { "grok-2",            LLM_CHAT_TEMPLATE_GROK_2            },80    { "pangu-embedded",    LLM_CHAT_TEMPLATE_PANGU_EMBED       },81    { "solar-open",        LLM_CHAT_TEMPLATE_SOLAR_OPEN        },82};83 84llm_chat_template llm_chat_template_from_str(const std::string & name) {85    return LLM_CHAT_TEMPLATES.at(name);86}87 88llm_chat_template llm_chat_detect_template(const std::string & tmpl) {89    try {90        return llm_chat_template_from_str(tmpl);91    } catch (const std::out_of_range &) {92        // ignore93    }94 95    auto tmpl_contains = [&tmpl](const char * haystack) -> bool {96        return tmpl.find(haystack) != std::string::npos;97    };98    if (tmpl_contains("<|im_start|>")) {99        return tmpl_contains("<|im_sep|>")100            ? LLM_CHAT_TEMPLATE_PHI_4101            : tmpl_contains("<end_of_utterance>")102                ? LLM_CHAT_TEMPLATE_SMOLVLM // SmolVLM uses <|im_start|> as BOS, but it is NOT chatml103                : LLM_CHAT_TEMPLATE_CHATML;104    } else if (tmpl.find("mistral") == 0 || tmpl_contains("[INST]")) {105        if (tmpl_contains("[SYSTEM_PROMPT]")) {106            return LLM_CHAT_TEMPLATE_MISTRAL_V7;107        } else if (108            // catches official 'v1' template109            tmpl_contains("' [INST] ' + system_message")110            // catches official 'v3' and 'v3-tekken' templates111            || tmpl_contains("[AVAILABLE_TOOLS]")112        ) {113            // Official mistral 'v1', 'v3' and 'v3-tekken' templates114            // See: https://github.com/mistralai/cookbook/blob/main/concept-deep-dive/tokenization/chat_templates.md115            // See: https://github.com/mistralai/cookbook/blob/main/concept-deep-dive/tokenization/templates.md116            if (tmpl_contains(" [INST]")) {117                return LLM_CHAT_TEMPLATE_MISTRAL_V1;118            } else if (tmpl_contains("\"[INST]\"")) {119                return LLM_CHAT_TEMPLATE_MISTRAL_V3_TEKKEN;120            }121            return LLM_CHAT_TEMPLATE_MISTRAL_V3;122        } else {123            // llama2 template and its variants124            // [variant] support system message125            // See: https://huggingface.co/blog/llama2#how-to-prompt-llama-2126            bool support_system_message = tmpl_contains("<<SYS>>");127            bool add_bos_inside_history = tmpl_contains("bos_token + '[INST]");128            bool strip_message = tmpl_contains("content.strip()");129            if (strip_message) {130                return LLM_CHAT_TEMPLATE_LLAMA_2_SYS_STRIP;131            } else if (add_bos_inside_history) {132                return LLM_CHAT_TEMPLATE_LLAMA_2_SYS_BOS;133            } else if (support_system_message) {134                return LLM_CHAT_TEMPLATE_LLAMA_2_SYS;135            } else {136                return LLM_CHAT_TEMPLATE_LLAMA_2;137            }138        }139    } else if (tmpl_contains("<|assistant|>") && tmpl_contains("<|end|>")) {140        return LLM_CHAT_TEMPLATE_PHI_3;141    } else if (tmpl_contains("[gMASK]<sop>")) {142        return LLM_CHAT_TEMPLATE_CHATGLM_4;143    } else if (tmpl_contains("<|assistant|>") && tmpl_contains("<|user|>")) {144        if (tmpl_contains("<|tool_declare|>")) {145            return LLM_CHAT_TEMPLATE_EXAONE_MOE;146        }147        return tmpl_contains("</s>") ? LLM_CHAT_TEMPLATE_FALCON_3 : LLM_CHAT_TEMPLATE_GLMEDGE;148    } else if (tmpl_contains("<|{{ item['role'] }}|>") && tmpl_contains("<|begin_of_image|>")) {149        return LLM_CHAT_TEMPLATE_GLMEDGE;150    } else if (tmpl_contains("<|user|>") && tmpl_contains("<|endoftext|>")) {151        return LLM_CHAT_TEMPLATE_ZEPHYR;152    } else if (tmpl_contains("bos_token + message['role']")) {153        return LLM_CHAT_TEMPLATE_MONARCH;154    } else if (tmpl_contains("<start_of_turn>")) {155        return LLM_CHAT_TEMPLATE_GEMMA;156    } else if (tmpl_contains("'\\n\\nAssistant: ' + eos_token")) {157        // OrionStarAI/Orion-14B-Chat158        return LLM_CHAT_TEMPLATE_ORION;159    } else if (tmpl_contains("GPT4 Correct ")) {160        // openchat/openchat-3.5-0106161        return LLM_CHAT_TEMPLATE_OPENCHAT;162    } else if (tmpl_contains("USER: ") && tmpl_contains("ASSISTANT: ")) {163        // eachadea/vicuna-13b-1.1 (and Orca variant)164        if (tmpl_contains("SYSTEM: ")) {165            return LLM_CHAT_TEMPLATE_VICUNA_ORCA;166        }167        return LLM_CHAT_TEMPLATE_VICUNA;168    } else if (tmpl_contains("### Instruction:") && tmpl_contains("<|EOT|>")) {169        // deepseek-ai/deepseek-coder-33b-instruct170        return LLM_CHAT_TEMPLATE_DEEPSEEK;171    } else if (tmpl_contains("<|START_OF_TURN_TOKEN|>") && tmpl_contains("<|USER_TOKEN|>")) {172        // CohereForAI/c4ai-command-r-plus173        return LLM_CHAT_TEMPLATE_COMMAND_R;174    } else if (tmpl_contains("<|start_header_id|>") && tmpl_contains("<|end_header_id|>")) {175        return LLM_CHAT_TEMPLATE_LLAMA_3;176    } else if (tmpl_contains("[gMASK]sop")) {177        // chatglm3-6b178        return LLM_CHAT_TEMPLATE_CHATGLM_3;179    } else if (tmpl_contains(LU8("<用户>"))) {180        // MiniCPM-3B-OpenHermes-2.5-v2-GGUF181        return LLM_CHAT_TEMPLATE_MINICPM;182    } else if (tmpl_contains("'Assistant: ' + message['content'] + eos_token")) {183        return LLM_CHAT_TEMPLATE_DEEPSEEK_2;184    } else if (tmpl_contains(LU8("<|Assistant|>")) && tmpl_contains(LU8("<|User|>")) && tmpl_contains(LU8("<|end▁of▁sentence|>"))) {185        return LLM_CHAT_TEMPLATE_DEEPSEEK_3;186    } else if (tmpl_contains("[|system|]") && tmpl_contains("[|assistant|]") && tmpl_contains("[|endofturn|]")) {187        if (tmpl_contains("[|tool|]")) {188            return LLM_CHAT_TEMPLATE_EXAONE_4;189        }190        // ref: https://huggingface.co/LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct/discussions/8#66bae61b1893d14ee8ed85bb191        // EXAONE-3.0-7.8B-Instruct192        return LLM_CHAT_TEMPLATE_EXAONE_3;193    } else if (tmpl_contains("rwkv-world") || tmpl_contains("{{- 'User: ' + message['content']|trim + '\\n\\n' -}}")) {194        return LLM_CHAT_TEMPLATE_RWKV_WORLD;195    } else if (tmpl_contains("<|start_of_role|>")) {196        if (tmpl_contains("<tool_call>") || tmpl_contains("<tools>")) {197            return LLM_CHAT_TEMPLATE_GRANITE_4_0;198        }199        return LLM_CHAT_TEMPLATE_GRANITE_3_X;200    } else if (tmpl_contains("message['role'] + additional_special_tokens[0] + message['content'] + additional_special_tokens[1]")) {201        return LLM_CHAT_TEMPLATE_GIGACHAT;202    } else if (tmpl_contains("<|role_start|>")) {203        return LLM_CHAT_TEMPLATE_MEGREZ;204    } else if (tmpl_contains(" Ассистент:")) {205        return LLM_CHAT_TEMPLATE_YANDEX;206    } else if (tmpl_contains("<role>ASSISTANT</role>") && tmpl_contains("'HUMAN'")) {207        return LLM_CHAT_TEMPLATE_BAILING;208    } else if (tmpl_contains("<role>ASSISTANT</role>") && tmpl_contains("\"HUMAN\"") && tmpl_contains("<think>")) {209        return LLM_CHAT_TEMPLATE_BAILING_THINK;210    } else if (tmpl_contains("<role>ASSISTANT</role>") && tmpl_contains("<role>HUMAN</role>") && tmpl_contains("<|role_end|>")) {211        return LLM_CHAT_TEMPLATE_BAILING2;212    } else if (tmpl_contains("<|header_start|>") && tmpl_contains("<|header_end|>")) {213        return LLM_CHAT_TEMPLATE_LLAMA4;214    } else if (tmpl_contains("<|endofuserprompt|>")) {215        return LLM_CHAT_TEMPLATE_DOTS1;216    } else if (tmpl_contains("<|extra_0|>") && tmpl_contains("<|extra_4|>")) {217        return LLM_CHAT_TEMPLATE_HUNYUAN_MOE;218    } else if (tmpl_contains("<|start|>") && tmpl_contains("<|channel|>")) {219        return LLM_CHAT_TEMPLATE_OPENAI_MOE;220    } else if (tmpl_contains("<|hy_Assistant|>") && tmpl_contains("<|hy_begin▁of▁sentence|>")) {221        return LLM_CHAT_TEMPLATE_HUNYUAN_OCR;222    } else if (tmpl_contains("<|hy_Assistant|>") && tmpl_contains("<|hy_place▁holder▁no▁3|>")) {223        return LLM_CHAT_TEMPLATE_HUNYUAN_DENSE;224    } else if (tmpl_contains("<|im_assistant|>assistant<|im_middle|>")) {225        return LLM_CHAT_TEMPLATE_KIMI_K2;226    } else if (tmpl_contains("<seed:bos>")) {227        return LLM_CHAT_TEMPLATE_SEED_OSS;228    } else if (tmpl_contains("'Assistant: '  + message['content'] + '<|separator|>")) {229        return LLM_CHAT_TEMPLATE_GROK_2;230    } else if (tmpl_contains(LU8("[unused9]系统:[unused10]"))) {231        return LLM_CHAT_TEMPLATE_PANGU_EMBED;232    } else if (tmpl_contains("<|begin|>") && tmpl_contains("<|end|>") && tmpl_contains("<|content|>")) {233        return LLM_CHAT_TEMPLATE_SOLAR_OPEN;234    }235    return LLM_CHAT_TEMPLATE_UNKNOWN;236}237 238// Simple version of "llama_apply_chat_template" that only works with strings239// This function uses heuristic checks to determine commonly used template. It is not a jinja parser.240int32_t llm_chat_apply_template(241    llm_chat_template tmpl,242    const std::vector<const llama_chat_message *> & chat,243    std::string & dest, bool add_ass) {244    // Taken from the research: https://github.com/ggml-org/llama.cpp/issues/5527245    std::stringstream ss;246    if (tmpl == LLM_CHAT_TEMPLATE_CHATML) {247        // chatml template248        for (auto message : chat) {249            ss << "<|im_start|>" << message->role << "\n" << message->content << "<|im_end|>\n";250        }251        if (add_ass) {252            ss << "<|im_start|>assistant\n";253        }254    } else if (tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V7 || tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V7_TEKKEN) {255        // Official mistral 'v7' template256        // See: https://huggingface.co/mistralai/Mistral-Large-Instruct-2411#basic-instruct-template-v7257        //      https://huggingface.co/mistralai/Mistral-Small-3.1-24B-Instruct-2503#basic-instruct-template-v7-tekken258        const char * trailing_space = tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V7 ? " " : "";259        for (auto message : chat) {260            std::string role(message->role);261            std::string content(message->content);262            if (role == "system") {263                ss << "[SYSTEM_PROMPT]" << trailing_space << content << "[/SYSTEM_PROMPT]";264            } else if (role == "user") {265                ss << "[INST]" << trailing_space << content << "[/INST]";266            } else {267                ss << trailing_space << content << "</s>";268            }269        }270    } else if (tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V1271            || tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V3272            || tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V3_TEKKEN) {273        // See: https://github.com/mistralai/cookbook/blob/main/concept-deep-dive/tokenization/chat_templates.md274        // See: https://github.com/mistralai/cookbook/blob/main/concept-deep-dive/tokenization/templates.md275        std::string leading_space = tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V1 ? " " : "";276        std::string trailing_space = tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V3_TEKKEN ? "" : " ";277        bool trim_assistant_message = tmpl == LLM_CHAT_TEMPLATE_MISTRAL_V3;278        bool is_inside_turn = false;279        for (auto message : chat) {280            if (!is_inside_turn) {281                ss << leading_space << "[INST]" << trailing_space;282                is_inside_turn = true;283            }284            std::string role(message->role);285            std::string content(message->content);286            if (role == "system") {287                ss << content << "\n\n";288            } else if (role == "user") {289                ss << content << leading_space << "[/INST]";290            } else {291                ss << trailing_space << (trim_assistant_message ? trim(content) : content) << "</s>";292                is_inside_turn = false;293            }294        }295    } else if (296            tmpl == LLM_CHAT_TEMPLATE_LLAMA_2297            || tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS298            || tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS_BOS299            || tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS_STRIP) {300        // llama2 template and its variants301        // [variant] support system message302        // See: https://huggingface.co/blog/llama2#how-to-prompt-llama-2303        bool support_system_message = tmpl != LLM_CHAT_TEMPLATE_LLAMA_2;304        // [variant] add BOS inside history305        bool add_bos_inside_history = tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS_BOS;306        // [variant] trim spaces from the input message307        bool strip_message = tmpl == LLM_CHAT_TEMPLATE_LLAMA_2_SYS_STRIP;308        // construct the prompt309        bool is_inside_turn = true; // skip BOS at the beginning310        ss << "[INST] ";311        for (auto message : chat) {312            std::string content = strip_message ? trim(message->content) : message->content;313            std::string role(message->role);314            if (!is_inside_turn) {315                is_inside_turn = true;316                ss << (add_bos_inside_history ? "<s>[INST] " : "[INST] ");317            }318            if (role == "system") {319                if (support_system_message) {320                    ss << "<<SYS>>\n" << content << "\n<</SYS>>\n\n";321                } else {322                    // if the model does not support system message, we still include it in the first message, but without <<SYS>>323                    ss << content << "\n";324                }325            } else if (role == "user") {326                ss << content << " [/INST]";327            } else {328                ss << content << "</s>";329                is_inside_turn = false;330            }331        }332    } else if (tmpl == LLM_CHAT_TEMPLATE_PHI_3) {333        // Phi 3334        for (auto message : chat) {335            std::string role(message->role);336            ss << "<|" << role << "|>\n" << message->content << "<|end|>\n";337        }338        if (add_ass) {339            ss << "<|assistant|>\n";340        }341    } else if (tmpl == LLM_CHAT_TEMPLATE_PHI_4) {342        // chatml template343        for (auto message : chat) {344            ss << "<|im_start|>" << message->role << "<|im_sep|>" << message->content << "<|im_end|>";345        }346        if (add_ass) {347            ss << "<|im_start|>assistant<|im_sep|>";348        }349    } else if (tmpl == LLM_CHAT_TEMPLATE_FALCON_3) {350        // Falcon 3351        for (auto message : chat) {352            std::string role(message->role);353            ss << "<|" << role << "|>\n" << message->content << "\n";354        }355        if (add_ass) {356            ss << "<|assistant|>\n";357        }358    } else if (tmpl == LLM_CHAT_TEMPLATE_ZEPHYR) {359        // zephyr template360        for (auto message : chat) {361            ss << "<|" << message->role << "|>" << "\n" << message->content << "<|endoftext|>\n";362        }363        if (add_ass) {364            ss << "<|assistant|>\n";365        }366    } else if (tmpl == LLM_CHAT_TEMPLATE_MONARCH) {367        // mlabonne/AlphaMonarch-7B template (the <s> is included inside history)368        for (auto message : chat) {369            std::string bos = (message == chat.front()) ? "" : "<s>"; // skip BOS for first message370            ss << bos << message->role << "\n" << message->content << "</s>\n";371        }372        if (add_ass) {373            ss << "<s>assistant\n";374        }375    } else if (tmpl == LLM_CHAT_TEMPLATE_GEMMA) {376        // google/gemma-7b-it377        std::string system_prompt = "";378        for (auto message : chat) {379            std::string role(message->role);380            if (role == "system") {381                // there is no system message for gemma, but we will merge it with user prompt, so nothing is broken382                system_prompt += trim(message->content);383                continue;384            }385            // in gemma, "assistant" is "model"386            role = role == "assistant" ? "model" : message->role;387            ss << "<start_of_turn>" << role << "\n";388            if (!system_prompt.empty() && role != "model") {389                ss << system_prompt << "\n\n";390                system_prompt = "";391            }392            ss << trim(message->content) << "<end_of_turn>\n";393        }394        if (add_ass) {395            ss << "<start_of_turn>model\n";396        }397    } else if (tmpl == LLM_CHAT_TEMPLATE_ORION) {398        // OrionStarAI/Orion-14B-Chat399        std::string system_prompt = "";400        for (auto message : chat) {401            std::string role(message->role);402            if (role == "system") {403                // there is no system message support, we will merge it with user prompt404                system_prompt += message->content;405                continue;406            } else if (role == "user") {407                ss << "Human: ";408                if (!system_prompt.empty()) {409                    ss << system_prompt << "\n\n";410                    system_prompt = "";411                }412                ss << message->content << "\n\nAssistant: </s>";413            } else {414                ss << message->content << "</s>";415            }416        }417    } else if (tmpl == LLM_CHAT_TEMPLATE_OPENCHAT) {418        // openchat/openchat-3.5-0106,419        for (auto message : chat) {420            std::string role(message->role);421            if (role == "system") {422                ss << message->content << "<|end_of_turn|>";423            } else {424                role[0] = toupper(role[0]);425                ss << "GPT4 Correct " << role << ": " << message->content << "<|end_of_turn|>";426            }427        }428        if (add_ass) {429            ss << "GPT4 Correct Assistant:";430        }431    } else if (tmpl == LLM_CHAT_TEMPLATE_VICUNA || tmpl == LLM_CHAT_TEMPLATE_VICUNA_ORCA) {432        // eachadea/vicuna-13b-1.1 (and Orca variant)433        for (auto message : chat) {434            std::string role(message->role);435            if (role == "system") {436                // Orca-Vicuna variant uses a system prefix437                if (tmpl == LLM_CHAT_TEMPLATE_VICUNA_ORCA) {438                    ss << "SYSTEM: " << message->content << "\n";439                } else {440                    ss << message->content << "\n\n";441                }442            } else if (role == "user") {443                ss << "USER: " << message->content << "\n";444            } else if (role == "assistant") {445                ss << "ASSISTANT: " << message->content << "</s>\n";446            }447        }448        if (add_ass) {449            ss << "ASSISTANT:";450        }451    } else if (tmpl == LLM_CHAT_TEMPLATE_DEEPSEEK) {452        // deepseek-ai/deepseek-coder-33b-instruct453        for (auto message : chat) {454            std::string role(message->role);455            if (role == "system") {456                ss << message->content;457            } else if (role == "user") {458                ss << "### Instruction:\n" << message->content << "\n";459            } else if (role == "assistant") {460                ss << "### Response:\n" << message->content << "\n<|EOT|>\n";461            }462        }463        if (add_ass) {464            ss << "### Response:\n";465        }466    } else if (tmpl == LLM_CHAT_TEMPLATE_COMMAND_R) {467        // CohereForAI/c4ai-command-r-plus468        for (auto message : chat) {469            std::string role(message->role);470            if (role == "system") {471                ss << "<|START_OF_TURN_TOKEN|><|SYSTEM_TOKEN|>" << trim(message->content) << "<|END_OF_TURN_TOKEN|>";472            } else if (role == "user") {473                ss << "<|START_OF_TURN_TOKEN|><|USER_TOKEN|>" << trim(message->content) << "<|END_OF_TURN_TOKEN|>";474            } else if (role == "assistant") {475                ss << "<|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|>" << trim(message->content) << "<|END_OF_TURN_TOKEN|>";476            }477        }478        if (add_ass) {479            ss << "<|START_OF_TURN_TOKEN|><|CHATBOT_TOKEN|>";480        }481    } else if (tmpl == LLM_CHAT_TEMPLATE_LLAMA_3) {482        // Llama 3483        for (auto message : chat) {484            std::string role(message->role);485            ss << "<|start_header_id|>" << role << "<|end_header_id|>\n\n" << trim(message->content) << "<|eot_id|>";486        }487        if (add_ass) {488            ss << "<|start_header_id|>assistant<|end_header_id|>\n\n";489        }490    } else if (tmpl == LLM_CHAT_TEMPLATE_CHATGLM_3) {491        // chatglm3-6b492        ss << "[gMASK]" << "sop";493        for (auto message : chat) {494            std::string role(message->role);495            ss << "<|" << role << "|>" << "\n " << message->content;496        }497        if (add_ass) {498            ss << "<|assistant|>";499        }500    } else if (tmpl == LLM_CHAT_TEMPLATE_CHATGLM_4) {501        ss << "[gMASK]" << "<sop>";502        for (auto message : chat) {503            std::string role(message->role);504            ss << "<|" << role << "|>" << "\n" << message->content;505        }506        if (add_ass) {507            ss << "<|assistant|>\n";508        }509    } else if (tmpl == LLM_CHAT_TEMPLATE_GLMEDGE) {510        for (auto message : chat) {511            std::string role(message->role);512            ss << "<|" << role << "|>" << "\n" << message->content;513        }514        if (add_ass) {515            ss << "<|assistant|>";516        }517    } else if (tmpl == LLM_CHAT_TEMPLATE_MINICPM) {518        // MiniCPM-3B-OpenHermes-2.5-v2-GGUF519        for (auto message : chat) {520            std::string role(message->role);521            if (role == "user") {522                ss << LU8("<用户>");523                ss << trim(message->content);524                ss << "<AI>";525            } else {526                ss << trim(message->content);527            }528        }529    } else if (tmpl == LLM_CHAT_TEMPLATE_DEEPSEEK_2) {530        // DeepSeek-V2531        for (auto message : chat) {532            std::string role(message->role);533            if (role == "system") {534                ss << message->content << "\n\n";535            } else if (role == "user") {536                ss << "User: " << message->content << "\n\n";537            } else if (role == "assistant") {538                ss << "Assistant: " << message->content << LU8("<|end▁of▁sentence|>");539            }540        }541        if (add_ass) {542            ss << "Assistant:";543        }544    } else if (tmpl == LLM_CHAT_TEMPLATE_DEEPSEEK_3) {545        // DeepSeek-V3546        for (auto message : chat) {547            std::string role(message->role);548            if (role == "system") {549                ss << message->content << "\n\n";550            } else if (role == "user") {551                ss << LU8("<|User|>") << message->content;552            } else if (role == "assistant") {553                ss << LU8("<|Assistant|>") << message->content << LU8("<|end▁of▁sentence|>");554            }555        }556        if (add_ass) {557            ss << LU8("<|Assistant|>");558        }559    } else if (tmpl == LLM_CHAT_TEMPLATE_DEEPSEEK_OCR) {560        for (auto message : chat) {561            // no template562            ss << message->content;563        }564    } else if (tmpl == LLM_CHAT_TEMPLATE_EXAONE_3) {565        // ref: https://huggingface.co/LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct/discussions/8#66bae61b1893d14ee8ed85bb566        // EXAONE-3.0-7.8B-Instruct567        for (auto message : chat) {568            std::string role(message->role);569            if (role == "system") {570                ss << "[|system|]" << trim(message->content) << "[|endofturn|]\n";571            } else if (role == "user") {572                ss << "[|user|]" << trim(message->content) << "\n";573            } else if (role == "assistant") {574                ss << "[|assistant|]" << trim(message->content) << "[|endofturn|]\n";575            }576        }577        if (add_ass) {578            ss << "[|assistant|]";579        }580    } else if (tmpl == LLM_CHAT_TEMPLATE_EXAONE_4) {581        for (auto message : chat) {582            std::string role(message->role);583            if (role == "system") {584                ss << "[|system|]" << trim(message->content) << "[|endofturn|]\n";585            } else if (role == "user") {586                ss << "[|user|]" << trim(message->content) << "\n";587            } else if (role == "assistant") {588                ss << "[|assistant|]" << trim(message->content) << "[|endofturn|]\n";589            } else if (role == "tool") {590                ss << "[|tool|]" << trim(message->content) << "[|endofturn|]\n";591            }592        }593        if (add_ass) {594            ss << "[|assistant|]";595        }596    } else if (tmpl == LLM_CHAT_TEMPLATE_EXAONE_MOE) {597        for (auto message : chat) {598            std::string role(message->role);599            if (role == "system") {600                ss << "<|system|>\n" << trim(message->content) << "<|endofturn|>\n";601            } else if (role == "user") {602                ss << "<|user|>\n" << trim(message->content) << "<|endofturn|>\n";603            } else if (role == "assistant") {604                ss << "<|assistant|>\n" << trim(message->content) << "<|endofturn|>\n";605            } else if (role == "tool") {606                ss << "<|tool|>\n" << trim(message->content) << "<|endofturn|>\n";607            }608        }609        if (add_ass) {610            ss << "<|assistant|>\n";611        }612    } else if (tmpl == LLM_CHAT_TEMPLATE_RWKV_WORLD) {613        // this template requires the model to have "\n\n" as EOT token614        for (size_t i = 0; i < chat.size(); i++) {615            std::string role(chat[i]->role);616            if (role == "system") {617                ss << "System: " << trim(chat[i]->content) << "\n\n";618            } else if (role == "user") {619                ss << "User: " << trim(chat[i]->content) << "\n\n";620                if (i == chat.size() - 1) {621                    ss << "Assistant:";622                }623            } else if (role == "assistant") {624                ss << "Assistant: " << trim(chat[i]->content) << "\n\n";625            }626        }627    } else if (tmpl == LLM_CHAT_TEMPLATE_GRANITE_3_X) {628        // IBM Granite 3.x template629        for (const auto & message : chat) {630            std::string role(message->role);631            ss << "<|start_of_role|>" << role << "<|end_of_role|>";632            if (role == "assistant_tool_call") {633                ss << "<|tool_call|>";634            }635            ss << message->content << "<|end_of_text|>\n";636        }637        if (add_ass) {638            ss << "<|start_of_role|>assistant<|end_of_role|>";639        }640    } else if (tmpl == LLM_CHAT_TEMPLATE_GRANITE_4_0) {641        // IBM Granite 4.0 template642        for (const auto & message : chat) {643            std::string role(message->role);644            if (role == "assistant_tool_call") {645                ss << "<|start_of_role|>assistant<|end_of_role|><|tool_call|>";646            } else {647                ss << "<|start_of_role|>" << role << "<|end_of_role|>";648            }649            ss << message->content << "<|end_of_text|>\n";650        }651        if (add_ass) {652            ss << "<|start_of_role|>assistant<|end_of_role|>";653        }654    } else if (tmpl == LLM_CHAT_TEMPLATE_GIGACHAT) {655        // GigaChat template656        bool has_system = !chat.empty() && std::string(chat[0]->role) == "system";657 658        // Handle system message if present659        if (has_system) {660            ss << "<s>" << chat[0]->content << "<|message_sep|>";661        } else {662            ss << "<s>";663        }664 665        // Process remaining messages666        for (size_t i = has_system ? 1 : 0; i < chat.size(); i++) {667            std::string role(chat[i]->role);668            if (role == "user") {669                ss << "user<|role_sep|>" << chat[i]->content << "<|message_sep|>"670                << "available functions<|role_sep|>[]<|message_sep|>";671            } else if (role == "assistant") {672                ss << "assistant<|role_sep|>" << chat[i]->content << "<|message_sep|>";673            }674        }675 676        // Add generation prompt if needed677        if (add_ass) {678            ss << "assistant<|role_sep|>";679        }680    }  else if (tmpl == LLM_CHAT_TEMPLATE_MEGREZ) {681        // Megrez template682        for (auto message : chat) {683            std::string role(message->role);684            ss << "<|role_start|>" << role << "<|role_end|>" << message->content << "<|turn_end|>";685        }686 687        if (add_ass) {688            ss << "<|role_start|>assistant<|role_end|>";689        }690    } else if (tmpl == LLM_CHAT_TEMPLATE_YANDEX) {691        // Yandex template ("\n\n" is defined as EOT token)692 693        for (size_t i = 0; i < chat.size(); i++) {694            std::string role(chat[i]->role);695            if (role == "user") {696                ss << " Пользователь: " << chat[i]->content << "\n\n";697            } else if (role == "assistant") {698                ss << " Ассистент: " << chat[i]->content << "\n\n";699            }700        }701 702        // Add generation prompt if needed703        if (add_ass) {704            ss << " Ассистент:[SEP]";705        }706    } else if (tmpl == LLM_CHAT_TEMPLATE_BAILING || tmpl == LLM_CHAT_TEMPLATE_BAILING_THINK) {707        // Bailing (Ling/Ring) template708        for (auto message : chat) {709            std::string role(message->role);710 711            if (role == "user") {712                role = "HUMAN";713            } else {714                std::transform(role.begin(), role.end(), role.begin(), ::toupper);715            }716 717            ss << "<role>" << role << "</role>" << message->content;718        }719 720        if (add_ass) {721            ss << "<role>ASSISTANT</role>";722 723            if (tmpl == LLM_CHAT_TEMPLATE_BAILING_THINK) {724                ss << "<think>";725            }726        }727    } else if (tmpl == LLM_CHAT_TEMPLATE_BAILING2) {728        // Bailing2 (Ling 2.0) template729        bool has_system = !chat.empty() && std::string(chat[0]->role) == "system";730 731        if (!has_system) {732            ss << "<role>SYSTEM</role>detailed thinking off<|role_end|>";733        }734 735        for (auto message : chat) {736            std::string role(message->role);737 738            if (role == "user") {739                role = "HUMAN";740            } else {741                std::transform(role.begin(), role.end(), role.begin(), ::toupper);742            }743 744            ss << "<role>" << role << "</role>" << message->content << "<|role_end|>";745        }746 747        if (add_ass) {748            ss << "<role>ASSISTANT</role>";749        }750    } else if (tmpl == LLM_CHAT_TEMPLATE_LLAMA4) {751        // Llama 4752        for (auto message : chat) {753            std::string role(message->role);754            ss << "<|header_start|>" << role << "<|header_end|>\n\n" << trim(message->content) << "<|eot|>";755        }756        if (add_ass) {757            ss << "<|header_start|>assistant<|header_end|>\n\n";758        }759    } else if (tmpl == LLM_CHAT_TEMPLATE_SMOLVLM) {760        // SmolVLM761        ss << "<|im_start|>"; // uses <|im_start|> as BOS, but the actual content is NOT chatml762        for (auto message : chat) {763            std::string role(message->role);764            if (role == "system") {765                ss << message->content << "\n\n";766            } else if (role == "user") {767                ss << "User: " << message->content << "<end_of_utterance>\n";768            } else {769                ss << "Assistant: " << message->content << "<end_of_utterance>\n";770            }771        }772        if (add_ass) {773            ss << "Assistant:";774        }775    } else if (tmpl == LLM_CHAT_TEMPLATE_DOTS1) {776        // dots.llm1.inst (DOTS1)777        for (auto message : chat) {778            std::string role(message->role);779            if (role == "system") {780                ss << "<|system|>" << message->content << "<|endofsystem|>";781            } else if (role == "user") {782                ss << "<|userprompt|>" << message->content << "<|endofuserprompt|>";783            } else {784                ss << "<|response|>" << message->content << "<|endofresponse|>";785            }786        }787        if (add_ass) {788            ss << "<|response|>";789        }790    } else if (tmpl == LLM_CHAT_TEMPLATE_HUNYUAN_MOE) {791        // tencent/Hunyuan-A13B-Instruct792        for (auto message : chat) {793            std::string role(message->role);794            if (role == "system") {795                ss << "<|startoftext|>" << message->content << "<|extra_4|>";796            } else if (role == "assistant") {797                ss << message->content << "<|eos|>";798            } else {799                ss << "<|startoftext|>" << message->content << "<|extra_0|>";800            }801        }802    } else if (tmpl == LLM_CHAT_TEMPLATE_OPENAI_MOE) {803        // OpenAI MoE (based on Harmony chat template)804        for (auto message : chat) {805            std::string role(message->role);806            ss << "<|start|>" << role << "<|message|>" << message->content;807            ss << (role == "assistant" ? "<|return|>" : "<|end|>");808        }809        if (add_ass) {810            ss << "<|start|>assistant";811        }812    } else if (tmpl == LLM_CHAT_TEMPLATE_HUNYUAN_DENSE) {813        // tencent/Hunyuan-4B-Instruct814        for (size_t i = 0; i < chat.size(); i++) {815            std::string role(chat[i]->role);816            if (i == 0) {817                if (role == "system") {818                    ss << chat[i]->content << "<|hy_place▁holder▁no▁3|>";819                }820            }821 822            if (role == "assistant") {823                ss << "<|hy_Assistant|>" << chat[i]->content << "<|hy_place▁holder▁no▁2|>";824            } else if (role == "user") {825                ss << "<|hy_User|>" << chat[i]->content << "<|hy_Assistant|>";826            }827        }828    } else if (tmpl == LLM_CHAT_TEMPLATE_HUNYUAN_OCR) {829        // tencent/HunyuanOCR830        ss << "<|hy_begin▁of▁sentence|>";831        for (size_t i = 0; i < chat.size(); i++) {832            std::string role(chat[i]->role);833            if (i == 0 && role == "system") {834                ss << chat[i]->content << "<|hy_place▁holder▁no▁3|>";835                continue;836            }837 838            if (role == "user") {839                ss << chat[i]->content << "<|hy_User|>";840            } else if (role == "assistant") {841                ss << chat[i]->content << "<|hy_Assistant|>";842            }843        }844    } else if (tmpl == LLM_CHAT_TEMPLATE_KIMI_K2) {845        // moonshotai/Kimi-K2-Instruct846        for (auto message : chat) {847            std::string role(message->role);848            if (role == "system") {849                ss << "<|im_system|>system<|im_middle|>";850            } else if (role == "user") {851                ss << "<|im_user|>user<|im_middle|>";852            } else if (role == "assistant") {853                ss << "<|im_assistant|>assistant<|im_middle|>";854            } else if (role == "tool") {855                ss << "<|im_system|>tool<|im_middle|>";856            }857 858            ss << message->content << "<|im_end|>";859        }860        if (add_ass) {861            ss << "<|im_assistant|>assistant<|im_middle|>";862        }863    } else if (tmpl == LLM_CHAT_TEMPLATE_SEED_OSS) {864        for (auto message: chat) {865            std::string role(message->role);866            ss << "<seed:bos>" << role << "\n" << (role == "assistant" ? trim(message->content) : message->content) << "<seed:eos>";867        }868        if (add_ass) {869            ss << "<seed:bos>assistant\n";870        }871    } else if (tmpl == LLM_CHAT_TEMPLATE_GROK_2) {872        for (auto message : chat) {873            std::string role(message->role);874            if (role == "system") {875                ss << "System: " << trim(message->content) << "<|separator|>\n\n";876            } else if (role == "user") {877                ss << "Human: " << trim(message->content) << "<|separator|>\n\n";878            } else if (role == "assistant") {879                ss << "Assistant: " << message->content << "<|separator|>\n\n";880            }881        }882        if (add_ass) {883            ss << "Assistant:";884        }885    }else if (tmpl == LLM_CHAT_TEMPLATE_PANGU_EMBED) {886        // [unused9]系统:xxx[unused10]887        // [unused9]用户:xxx[unused10]888        // [unused9]助手:xxx[unused10]889        // ...890        for (size_t i = 0; i < chat.size(); ++i) {891            const auto & msg = chat[i];892            const std::string & role = msg->role;893            const std::string & content = msg->content;894 895            if (i == 0 && role != "system") {896                ss << "[unused9]系统:[unused10]";897            }898 899            if (role == "system") {900                ss << "[unused9]系统:" << content << "[unused10]";901            } else if (role == "user") {902                ss << "[unused9]用户:" << content << "[unused10]";903            } else if (role == "assistant") {904                ss << "[unused9]助手:" << content << "[unused10]";905            } else if (role == "tool") {906                ss << "[unused9]工具:" << content << "[unused10]";907            } else if (role == "function") {908                ss << "[unused9]方法:" << content << "[unused10]";909            }910        }911        if (add_ass) {912            ss << "[unused9]助手:";913        }914    } else if (tmpl == LLM_CHAT_TEMPLATE_SOLAR_OPEN) {915        for (auto message : chat) {916            std::string role(message->role);917            ss << "<|begin|>" << role << "<|content|>" << message->content << "<|end|>";918        }919        if (add_ass) {920            ss << "<|begin|>assistant";921        }922    } else {923        // template not supported924        return -1;925    }926    dest = ss.str();927    return dest.size();928}929 930// public interface931 932int32_t llama_chat_builtin_templates(const char ** output, size_t len) {933    auto it = LLM_CHAT_TEMPLATES.begin();934    for (size_t i = 0; i < std::min(len, LLM_CHAT_TEMPLATES.size()); i++) {935        output[i] = it->first.c_str();936        std::advance(it, 1);937    }938    return (int32_t) LLM_CHAT_TEMPLATES.size();939}940