Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03k
1#include "../src/llama-grammar.h"2#include "chat-auto-parser.h"3#include "chat.h"4#include "common.h"5#include "gguf.h"6#include "jinja/runtime.h"7#include "log.h"8#include "nlohmann/json.hpp"9#include "peg-parser.h"10 11#include <fstream>12#include <iterator>13#include <numeric>14#include <optional>15#include <sstream>16#include <string>17 18using json = nlohmann::ordered_json;19 20enum class output_mode {21 ANALYSIS, // Only output analysis results (default)22 TEMPLATE, // Only output rendered template23 BOTH // Output both24};25 26enum class input_message_type {27 NONE, // Don't render any message scenarios (only analysis)28 CONTENT_ONLY, // Simple assistant message with content29 REASONING_CONTENT, // Message with reasoning_content + content30 TOOL_CALL_ONLY, // Message with tool_calls only31 CONTENT_TOOL_CALL, // Message with content + tool_calls32 REASONING_TOOL_CALL, // Message with reasoning_content + tool_calls33 CONTENT_FAKE_TOOL_CALL, // Message with content but no actual tool_calls (for testing)34 ALL // Render all scenarios35};36 37struct debug_options {38 std::string template_path;39 bool with_tools = true;40 bool generation_prompt = true;41 bool enable_reasoning = true;42 bool debug_jinja = false;43 bool force_tool_call = false;44 bool parallel_tool_calls = true;45 output_mode mode = output_mode::BOTH;46 input_message_type input_message = input_message_type::NONE;47};48 49static std::string read_file(const std::string & path) {50 std::ifstream fin(path, std::ios::binary);51 if (!fin.is_open()) {52 throw std::runtime_error("Could not open file: " + path);53 }54 std::ostringstream buf;55 buf << fin.rdbuf();56 return buf.str();57}58 59static std::string read_gguf_chat_template(const std::string & path) {60 struct gguf_init_params params = { /*no_alloc =*/true, // We only need metadata, not tensor data61 /*ctx=*/nullptr };62 63 struct gguf_context * ctx = gguf_init_from_file(path.c_str(), params);64 if (ctx == nullptr) {65 throw std::runtime_error("Could not open GGUF file: " + path);66 }67 68 const char * key = "tokenizer.chat_template";69 int64_t key_id = gguf_find_key(ctx, key);70 71 if (key_id == -1) {72 gguf_free(ctx);73 throw std::runtime_error("GGUF file does not contain chat template key: " + std::string(key));74 }75 76 const char * template_str = gguf_get_val_str(ctx, key_id);77 if (template_str == nullptr) {78 gguf_free(ctx);79 throw std::runtime_error("GGUF file contains chat template key but value is null");80 }81 82 std::string result = template_str;83 gguf_free(ctx);84 return result;85}86 87static void print_usage(const char * program_name) {88 LOG_ERR("Usage: %s <template_or_gguf_path> [options]\n", program_name);89 LOG_ERR("\nOptions:\n");90 LOG_ERR(" --no-tools Disable tool definitions\n");91 LOG_ERR(" --force-tool-call Set tool calls to forced\n");92 LOG_ERR(" --parallel-tool-calls=0|1 Set parallel_tool_calls (default: 1)\n");93 LOG_ERR(" --generation-prompt=0|1 Set add_generation_prompt (default: 1)\n");94 LOG_ERR(" --enable-reasoning=0|1 Enable reasoning parsing (default: 1)\n");95 LOG_ERR(" --output=MODE Output mode: analysis, template, both (default: both)\n");96 LOG_ERR(" --debug-jinja Enable Jinja fine-grained debug\n");97 LOG_ERR(" --input-message=TYPE Message type to render:\n");98 LOG_ERR(" content_only, reasoning_content, tool_call_only,\n");99 LOG_ERR(" content_tool_call, reasoning_tool_call,\n");100 LOG_ERR(" content_fake_tool_call, all\n");101 LOG_ERR("\nExamples:\n");102 LOG_ERR(" %s template.jinja --input-message=all --generation-prompt=1\n", program_name);103 LOG_ERR(" %s template.jinja --output=template --input-message=tool_call_only\n", program_name);104}105 106static bool parse_bool_option(const std::string & value) {107 return value == "1" || value == "true" || value == "yes";108}109 110static bool parse_options(int argc, char ** argv, debug_options & opts) {111 if (argc < 2) {112 print_usage(argv[0]);113 return false;114 }115 116 opts.template_path = argv[1];117 118 for (int i = 2; i < argc; ++i) {119 std::string arg = argv[i];120 121 if (arg == "--force-tool-call") {122 opts.force_tool_call = true;123 } else if (arg == "--debug-jinja") {124 opts.debug_jinja = true;125 } else if (arg == "--no-tools") {126 opts.with_tools = false;127 } else if (arg.rfind("--parallel-tool-calls=", 0) == 0) {128 opts.parallel_tool_calls = parse_bool_option(arg.substr(22));129 } else if (arg.rfind("--generation-prompt=", 0) == 0) {130 opts.generation_prompt = parse_bool_option(arg.substr(20));131 } else if (arg.rfind("--enable-reasoning=", 0) == 0) {132 opts.enable_reasoning = parse_bool_option(arg.substr(19));133 } else if (arg.rfind("--output=", 0) == 0) {134 std::string mode = arg.substr(9);135 if (mode == "analysis") {136 opts.mode = output_mode::ANALYSIS;137 } else if (mode == "template") {138 opts.mode = output_mode::TEMPLATE;139 } else if (mode == "both") {140 opts.mode = output_mode::BOTH;141 } else {142 LOG_ERR("Unknown output mode: %s\n", mode.c_str());143 return false;144 }145 } else if (arg.rfind("--input-message=", 0) == 0) {146 std::string type = arg.substr(16);147 if (type == "content_only") {148 opts.input_message = input_message_type::CONTENT_ONLY;149 } else if (type == "reasoning_content") {150 opts.input_message = input_message_type::REASONING_CONTENT;151 } else if (type == "tool_call_only") {152 opts.input_message = input_message_type::TOOL_CALL_ONLY;153 } else if (type == "content_tool_call") {154 opts.input_message = input_message_type::CONTENT_TOOL_CALL;155 } else if (type == "reasoning_tool_call") {156 opts.input_message = input_message_type::REASONING_TOOL_CALL;157 } else if (type == "content_fake_tool_call") {158 opts.input_message = input_message_type::CONTENT_FAKE_TOOL_CALL;159 } else if (type == "all") {160 opts.input_message = input_message_type::ALL;161 } else {162 LOG_ERR("Unknown input message type: %s\n", type.c_str());163 return false;164 }165 } else {166 LOG_ERR("Unknown option: %s\n", arg.c_str());167 print_usage(argv[0]);168 return false;169 }170 }171 172 return true;173}174 175static json build_user_message() {176 return json{177 { "role", "user" },178 { "content", "Hello, please help me with a task." }179 };180}181 182static json build_content_only_message() {183 return json{184 { "role", "assistant" },185 { "content", "Hello! I'm here to help you with your task." }186 };187}188 189static json build_reasoning_content_message() {190 return json{191 { "role", "assistant" },192 { "content", "Hello! I'm here to help you with your task." },193 { "reasoning_content", "The user is greeting me and asking for help. I should respond politely." }194 };195}196 197static json build_tool_call_only_message() {198 return json{199 { "role", "assistant" },200 { "content", nullptr },201 { "tool_calls",202 json::array({ json{203 { "type", "function" },204 { "function", json{ { "name", "test_function_name" },205 { "arguments", json::object({ { "param1", "value1" }, { "param2", "value2" } }) } } },206 { "id", "123456789" } } }) }207 };208}209 210static json build_content_tool_call_message() {211 return json{212 { "role", "assistant" },213 { "content", "I'll help you by calling a function." },214 { "tool_calls",215 json::array({ json{216 { "type", "function" },217 { "function",218 json{ { "name", "test_function_name" },219 { "arguments", json::object({ { "param1", "value1" }, { "param2", "value2" } }) } } } } }) }220 };221}222 223static json build_reasoning_tool_call_message() {224 return json{225 { "role", "assistant" },226 { "content", nullptr },227 { "reasoning_content", "I need to call a function to help with this task." },228 { "tool_calls",229 json::array({ json{230 { "type", "function" },231 { "function",232 json{ { "name", "test_function_name" },233 { "arguments", json::object({ { "param1", "value1" }, { "param2", "value2" } }) } } } } }) }234 };235}236 237static json build_content_fake_tool_call_message() {238 // This message has content but NO tool_calls field239 // It's used to test if a template renders tool definitions but not tool calls240 return json{241 { "role", "assistant" },242 { "content", "I'll help you by calling a function." }243 };244}245 246static json build_tools_definition() {247 json parameters_schema = json::object();248 parameters_schema["type"] = "object";249 parameters_schema["properties"] = json::object();250 parameters_schema["properties"]["param1"] = json::object({251 { "type", "string" },252 { "description", "First parameter" }253 });254 parameters_schema["properties"]["param2"] = json::object({255 { "type", "string" },256 { "description", "Second parameter" }257 });258 parameters_schema["required"] = json::array({ "param1" });259 260 return json::array({261 json{ { "type", "function" },262 { "function", json{ { "name", "test_function_name" },263 { "description", "A test function for debugging" },264 { "parameters", parameters_schema } } } }265 });266}267 268static void render_scenario(const common_chat_template & tmpl,269 const std::string & scenario_name,270 const json & messages,271 const json & tools,272 bool add_generation_prompt,273 bool enable_thinking) {274 LOG_ERR("\n=== Scenario: %s ===\n", scenario_name.c_str());275 LOG_ERR("add_generation_prompt: %s, enable_thinking: %s\n", add_generation_prompt ? "true" : "false",276 enable_thinking ? "true" : "false");277 278 // When add_generation_prompt is true, add a trailing user message to trigger the prompt279 json final_messages = messages;280 if (add_generation_prompt && !messages.empty() && messages.back().value("role", "") == "assistant") {281 final_messages.push_back(json{282 { "role", "user" },283 { "content", "Now please continue with another response." }284 });285 }286 287 LOG_ERR("Messages:\n%s\n", final_messages.dump(2).c_str());288 289 try {290 autoparser::generation_params inputs;291 inputs.messages = final_messages;292 inputs.add_generation_prompt = add_generation_prompt;293 inputs.extra_context["enable_thinking"] = enable_thinking;294 295 if (!tools.is_null() && tools.is_array() && !tools.empty()) {296 inputs.tools = tools;297 }298 299 std::string output = common_chat_template_direct_apply(tmpl, inputs);300 301 LOG_ERR("\n--- Rendered Output ---\n");302 LOG_ERR("%s\n", output.c_str());303 LOG_ERR("--- End Output (length: %zu) ---\n", output.length());304 } catch (const std::exception & e) {305 LOG_ERR("Rendering failed: %s\n", e.what());306 }307}308 309static void render_all_scenarios(const common_chat_template & tmpl,310 const json & tools,311 bool add_generation_prompt,312 bool enable_thinking,313 input_message_type message_type) {314 json user_msg = build_user_message();315 316 auto render_if = [&](input_message_type type, const std::string & name, const json & assistant_msg) {317 if (message_type == input_message_type::ALL || message_type == type) {318 json messages = json::array({ user_msg, assistant_msg });319 render_scenario(tmpl, name, messages, tools, add_generation_prompt, enable_thinking);320 }321 };322 323 render_if(input_message_type::CONTENT_ONLY, "content_only", build_content_only_message());324 render_if(input_message_type::REASONING_CONTENT, "reasoning_content", build_reasoning_content_message());325 render_if(input_message_type::TOOL_CALL_ONLY, "tool_call_only", build_tool_call_only_message());326 render_if(input_message_type::CONTENT_TOOL_CALL, "content_tool_call", build_content_tool_call_message());327 render_if(input_message_type::REASONING_TOOL_CALL, "reasoning_tool_call", build_reasoning_tool_call_message());328 render_if(input_message_type::CONTENT_FAKE_TOOL_CALL, "content_fake_tool_call",329 build_content_fake_tool_call_message());330 331 // Also render with add_generation_prompt=true to show the prompt ending332 if (message_type == input_message_type::ALL) {333 LOG_ERR("\n\n=== Generation Prompt Scenarios (add_generation_prompt=true) ===\n");334 335 json prompt_messages = json::array({ user_msg });336 render_scenario(tmpl, "generation_prompt_only", prompt_messages, tools, true, enable_thinking);337 338 // With enable_thinking toggled339 render_scenario(tmpl, "generation_prompt_thinking_disabled", prompt_messages, tools, true, false);340 }341}342 343static autoparser::generation_params prepare_params(const debug_options & opts, const json & tools) {344 autoparser::generation_params params;345 params.messages = json::array({ build_user_message() });346 params.reasoning_format = opts.enable_reasoning ? COMMON_REASONING_FORMAT_DEEPSEEK : COMMON_REASONING_FORMAT_NONE;347 params.enable_thinking = opts.enable_reasoning;348 params.add_generation_prompt = opts.generation_prompt;349 350 if (opts.with_tools) {351 params.tools = tools;352 params.tool_choice = opts.force_tool_call ? COMMON_CHAT_TOOL_CHOICE_REQUIRED : COMMON_CHAT_TOOL_CHOICE_AUTO;353 } else {354 params.tools = json();355 params.tool_choice = COMMON_CHAT_TOOL_CHOICE_NONE;356 }357 params.parallel_tool_calls = opts.parallel_tool_calls;358 return params;359}360 361int main(int argc, char ** argv) {362 // Set log level to most verbose to capture all debug output363 common_log_set_verbosity_thold(99);364 365 debug_options opts;366 if (!parse_options(argc, argv, opts)) {367 return 1;368 }369 370 if (opts.debug_jinja || std::getenv("LLAMA_DEBUG_JINJA") != nullptr) {371 jinja::enable_debug(true);372 }373 374 std::string template_source;375 try {376 // Check if the file is a GGUF file377 if (opts.template_path.size() >= 5 &&378 opts.template_path.compare(opts.template_path.size() - 5, 5, ".gguf") == 0) {379 template_source = read_gguf_chat_template(opts.template_path);380 } else {381 template_source = read_file(opts.template_path);382 }383 } catch (const std::exception & e) {384 LOG_ERR("Error reading template: %s\n", e.what());385 return 1;386 }387 388 LOG_ERR("Analyzing template: %s\n", opts.template_path.c_str());389 LOG_ERR("Options: with_tools=%s, generation_prompt=%s, enable_reasoning=%s\n", opts.with_tools ? "true" : "false",390 opts.generation_prompt ? "true" : "false", opts.enable_reasoning ? "true" : "false");391 392 try {393 common_chat_template chat_template(template_source, "", "");394 395 json tools = opts.with_tools ? build_tools_definition() : json();396 397 autoparser::generation_params params = prepare_params(opts, tools);398 common_chat_params parser_data;399 if (std::optional<common_chat_params> spec_tmpl =400 common_chat_try_specialized_template(chat_template, template_source, params)) {401 LOG_ERR("\n");402 LOG_ERR("This template uses a specialized parser, analysis results will not be available.\n");403 parser_data = *spec_tmpl;404 } else {405 // Render template scenarios if requested406 if (opts.input_message != input_message_type::NONE &&407 (opts.mode == output_mode::TEMPLATE || opts.mode == output_mode::BOTH)) {408 LOG_ERR("\n");409 LOG_ERR("================================================================================\n");410 LOG_ERR(" TEMPLATE RENDERING OUTPUT\n");411 LOG_ERR("================================================================================\n");412 413 render_all_scenarios(chat_template, tools, opts.generation_prompt, opts.enable_reasoning,414 opts.input_message);415 }416 417 // Output analysis if requested418 if (opts.mode == output_mode::ANALYSIS || opts.mode == output_mode::BOTH) {419 LOG_ERR("\n");420 LOG_ERR("================================================================================\n");421 LOG_ERR(" TEMPLATE ANALYSIS\n");422 LOG_ERR("================================================================================\n");423 424 autoparser::autoparser analysis;425 analysis.analyze_template(chat_template);426 427 // Generate Parser428 parser_data = autoparser::peg_generator::generate_parser(chat_template, params, analysis);429 }430 }431 432 if (!std::empty(parser_data.parser)) {433 LOG_ERR("\n=== Generated Parser ===\n");434 common_peg_arena arena;435 arena.load(parser_data.parser);436 LOG_ERR("%s\n", arena.dump(arena.root()).c_str());437 438 LOG_ERR("\n=== Generated Grammar ===\n");439 LOG_ERR("%s\n", parser_data.grammar.c_str());440 441 LOG_ERR("\n=== Generated Lazy Grammar ===\n");442 LOG_ERR("%d\n", parser_data.grammar_lazy);443 444 LOG_ERR("\n=== Generated Grammar Triggers ===\n");445 for (const common_grammar_trigger & cgt : parser_data.grammar_triggers) {446 LOG_ERR("Token: %d | Type: %d | Value: %s\n", cgt.token, cgt.type, cgt.value.c_str());447 }448 449 LOG_ERR("\n=== Preserved Tokens ===\n");450 for (const std::string & token : parser_data.preserved_tokens) {451 LOG_ERR(" '%s'\n", token.c_str());452 }453 454 if (!parser_data.grammar.empty()) {455 LOG_ERR("\n=== Verifying created grammar ===\n");456 auto * grammar = llama_grammar_init_impl(nullptr, parser_data.grammar.c_str(), "root",457 parser_data.grammar_lazy, nullptr, 0, nullptr, 0);458 if (grammar != nullptr) {459 LOG_ERR("\n=== Grammar successfully created ===\n");460 }461 }462 }463 } catch (const std::exception & e) {464 LOG_ERR("Analysis failed: %s\n", e.what());465 return 1;466 }467 468 return 0;469}470 