Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3kdownloads
chat-auto-parser-helpers.cpp367 linesDownload Raw Back to common
1#include "chat-auto-parser-helpers.h"2 3#include "chat-auto-parser.h"4#include "chat-peg-parser.h"5#include "chat.h"6#include "log.h"7#include "nlohmann/json.hpp"8#include "peg-parser.h"9 10#include <cctype>11#include <numeric>12 13using json = nlohmann::ordered_json;14 15std::string trim_whitespace(const std::string & str) {16    size_t start = 0;17    while (start < str.length() && std::isspace(static_cast<unsigned char>(str[start]))) {18        start++;19    }20 21    if (start == str.length()) {22        return "";23    }24 25    size_t end = str.length() - 1;26    while (end > start && std::isspace(static_cast<unsigned char>(str[end]))) {27        end--;28    }29 30    return str.substr(start, end - start + 1);31}32 33std::string trim_leading_whitespace(const std::string & str) {34    size_t start = 0;35    while (start < str.length() && std::isspace(static_cast<unsigned char>(str[start]))) {36        start++;37    }38 39    return str.substr(start);40}41 42std::string trim_trailing_whitespace(const std::string & str) {43    if (str.empty()) {44        return "";45    }46 47    size_t end = str.length() - 1;48    while (end > 0 && std::isspace(static_cast<unsigned char>(str[end]))) {49        end--;50    }51 52    // If first char is also whitespace, return empty string53    if (end == 0 && std::isspace(static_cast<unsigned char>(str[0]))) {54        return "";55    }56 57    return str.substr(0, end + 1);58}59 60std::string trim_trailing_newlines(const std::string & str) {61    size_t end = str.length();62    while (end > 0 && str[end - 1] == '\n') {63        end--;64    }65 66    return str.substr(0, end);67}68 69static size_t common_prefix_len(const std::string & left, const std::string & right) {70    size_t prefix_len = 0;71    size_t min_len    = std::min(left.length(), right.length());72    while (prefix_len < min_len && left[prefix_len] == right[prefix_len]) {73        prefix_len++;74    }75    return prefix_len;76}77 78static size_t common_suffix_len(const std::string & left, const std::string & right) {79    size_t suffix_len = 0;80    size_t min_len    = std::min(left.length(), right.length());81    while (suffix_len < min_len && left[left.length() - 1 - suffix_len] == right[right.length() - 1 - suffix_len]) {82        suffix_len++;83    }84    return suffix_len;85}86 87diff_split calculate_diff_split(const std::string & left, const std::string & right) {88    diff_split result;89 90    auto left_seg = segmentize_markers(left);91    auto right_seg = segmentize_markers(right);92 93    if (left_seg.empty()) {94        result.right = right;95        return result;96    }97    if (right_seg.empty()) {98        result.left = left;99        return result;100    }101 102    auto left_start = left_seg.begin();103    auto left_end = --left_seg.end();104    auto right_start = right_seg.begin();105    auto right_end = --right_seg.end();106 107    auto test = [&] () {108        return left_start != left_end && right_start != right_end;109    };110 111    bool left_fully_consumed = false;112    bool right_fully_consumed = false;113 114    while (test()) {115        bool advanced = false;116        if (*left_start == *right_start) {117            result.prefix.append(left_start->value);118            left_start++;119            right_start++;120            advanced = true;121        }122        if (*left_end == *right_end) {123            result.suffix = left_end->value + result.suffix;124            if (left_start != left_end) {125                left_end--;126            } else {127                left_fully_consumed = true;128            }129            if (right_start != right_end) {130                right_end--;131            } else {132                right_fully_consumed = true;133            }134            advanced = true;135        }136        if (!advanced) {137            break;138        }139    }140 141    if (left_start == left_end && right_start != right_end) {142        if (*left_start == *right_end) {143            result.suffix = right_end->value + result.suffix;144            right_end--;145            left_fully_consumed = true;146        } else if (*left_start == *right_start) {147            result.prefix.append(right_start->value);148            right_start++;149            left_fully_consumed = true;150        }151    } else if (right_start == right_end && left_start != left_end) {152        if (*left_end == *right_start) {153            result.suffix = left_end->value + result.suffix;154            left_end--;155            right_fully_consumed = true;156        } else if (*left_start == *right_start) {157            result.prefix.append(left_start->value);158            left_start++;159            right_fully_consumed = true;160        }161    } else if (left_start == left_end && right_start == right_end && *left_start == *right_start && left_start->type == segment_type::MARKER) {162        result.prefix.append(right_start->value);163        left_fully_consumed = true;164        right_fully_consumed = true;165    }166 167    auto eat_segment = [](std::string str, const segment & seg) -> std::string { return std::move(str) + seg.value; };168 169    bool can_have_text_suffix = left_end->type == segment_type::TEXT && right_end->type == segment_type::TEXT;170    bool can_have_text_prefix = right_start->type == segment_type::TEXT && left_start->type == segment_type::TEXT;171 172    std::string remainder_left = std::accumulate(left_start, left_fully_consumed ? left_end : ++left_end, std::string(), eat_segment);173    std::string remainder_right = std::accumulate(right_start, right_fully_consumed ? right_end : ++right_end, std::string(), eat_segment);174 175    size_t suffix_len = can_have_text_suffix ? common_suffix_len(remainder_left, remainder_right) : 0;176    // avoid overlaps between prefix and suffix177    size_t prefix_len = can_have_text_prefix ? common_prefix_len(remainder_left.substr(0, remainder_left.size() - suffix_len),178        remainder_right.substr(0, remainder_right.size() - suffix_len)) : 0;179 180    result.prefix.append(remainder_left.substr(0, prefix_len));181    result.suffix = remainder_left.substr(remainder_left.length() - suffix_len, suffix_len) + result.suffix;182    result.left = remainder_left.substr(prefix_len, remainder_left.length() - prefix_len - suffix_len);183    result.right = remainder_right.substr(prefix_len, remainder_right.length() - prefix_len - suffix_len);184 185    if (result.left == "" && result.right == "") {186        // degenerate case, no diff187        result.prefix = left;188        result.suffix = "";189        // pick prefix = all as representation190    }191 192    // When left has no unique content (result.left is empty), left is entirely193    // shared with right. The simultaneous prefix/suffix segment matching can194    // incorrectly consume trailing segments of left as suffix when those same195    // segments also appear at the end of right (e.g. "\n" at the end of both196    // the shared content and the generation prompt). This rotates the diff.197    // Fix: if left is a prefix of right, enforce that directly.198    if (result.left.empty() && !result.right.empty() &&199            left.size() <= right.size() &&200            right.substr(0, left.size()) == left) {201        result.prefix = left;202        result.suffix = "";203        result.right  = right.substr(left.size());204    }205 206    return result;207}208 209// Returns the prefix of `full` up until the first occurrence of the common prefix of `left` and `right`210std::string until_common_prefix(const std::string & full, const std::string & left, const std::string & right) {211    // Find the common prefix of left and right212    size_t common_prefix_len = 0;213    size_t min_len           = std::min(left.length(), right.length());214    while (common_prefix_len < min_len && left[common_prefix_len] == right[common_prefix_len]) {215        common_prefix_len++;216    }217 218    // If there's no common prefix, return empty string219    if (common_prefix_len == 0) {220        return "";221    }222 223    // Find the common prefix in the full string224    std::string common_prefix = left.substr(0, common_prefix_len);225    size_t      pos           = full.find(common_prefix);226 227    // If not found, return empty string228    if (pos == std::string::npos) {229        return "";230    }231 232    // Return everything before the common prefix233    return full.substr(0, pos);234}235 236// Returns the suffix of `full` after the last occurrence of the common suffix of `left` and `right`237std::string after_common_suffix(const std::string & full, const std::string & left, const std::string & right) {238    // Find the common suffix of left and right (compare from the end)239    size_t common_suffix_len = 0;240    size_t min_len           = std::min(left.length(), right.length());241    while (common_suffix_len < min_len &&242           left[left.length() - 1 - common_suffix_len] == right[right.length() - 1 - common_suffix_len]) {243        common_suffix_len++;244    }245 246    // If there's no common suffix, return empty string247    if (common_suffix_len == 0) {248        return "";249    }250 251    // Extract the common suffix252    std::string common_suffix = left.substr(left.length() - common_suffix_len);253 254    // Find the last occurrence of the common suffix in the full string255    size_t pos = full.rfind(common_suffix);256 257    // If not found, return empty string258    if (pos == std::string::npos) {259        return "";260    }261 262    // Return everything after the common suffix263    return full.substr(pos + common_suffix_len);264}265 266// TODO: segmentize will treat a JSON array inside tags as a tag: <calls>[{ "fun": { ... } }]</calls> will be three markers267// not too worried about that because it hasn't turned out as a problem anywhere, but noting here in case it will268// Might have to put some restrictions on tag contents as well (like "no { }")269std::vector<segment> segmentize_markers(const std::string & text) {270    std::vector<segment> retval;271    bool in_marker = false;272    char marker_opener = '\0';273 274    auto is_marker_opener = [](char c) -> bool { return c == '<' || c == '['; };275    auto is_marker_closer = [](char op, char c) -> bool { return (op == '<' && c == '>') || (op == '[' && c == ']'); };276 277    size_t last_border = 0;278 279    for (size_t cur_pos = 0; cur_pos < text.length(); cur_pos++) {280        if (!in_marker && is_marker_opener(text[cur_pos])) {281            if (last_border < cur_pos) {282                retval.push_back(segment(segment_type::TEXT, text.substr(last_border, cur_pos - last_border)));283            }284            last_border = cur_pos;285            in_marker = true;286            marker_opener = text[cur_pos];287        } else if (in_marker && is_marker_closer(marker_opener, text[cur_pos])) {288            // no need to check because last_border will always be smaller289                retval.push_back(segment(segment_type::MARKER, text.substr(last_border, cur_pos - last_border + 1)));290            last_border = cur_pos + 1;291            in_marker = false;292            marker_opener = '\0';293        }294    }295    if (last_border < text.length()) {296            retval.push_back(segment(segment_type::TEXT, text.substr(last_border)));297    }298    return retval;299}300 301std::vector<segment> prune_whitespace_segments(const std::vector<segment> & segments) {302    std::vector<segment> result;303    for (const auto & seg : segments) {304        if (!trim_whitespace(seg.value).empty()) {305            result.push_back(seg);306        }307    }308    return result;309}310 311namespace autoparser {312 313static const std::string ERR_TMPL = "#**ERROR**#";314 315std::string apply_template(const common_chat_template & tmpl, const template_params & params) {316    generation_params tmpl_params;317    tmpl_params.messages              = params.messages;318    tmpl_params.tools                 = params.tools;319    tmpl_params.add_generation_prompt = params.add_generation_prompt;320    tmpl_params.enable_thinking       = params.enable_thinking;321 322    if (params.extra_context) {323        tmpl_params.extra_context = *params.extra_context;324    }325    tmpl_params.extra_context["enable_thinking"] = params.enable_thinking;326 327    try {328        return common_chat_template_direct_apply(tmpl, tmpl_params);329    } catch (const std::exception & e) {330        LOG_DBG("Template application failed: %s\n", e.what());331        return ERR_TMPL;332    }333}334 335std::optional<compare_variants_result> compare_variants(336    const common_chat_template &                   tmpl,337    const template_params &                        params_A,338    const std::function<void(template_params &)> & params_modifier) {339    // Create variant B by copying A340    template_params params_B = params_A;341 342    // Apply modifier to create variant B343    if (params_modifier) {344        params_modifier(params_B);345    }346 347    // Apply template to both variants348    std::string output_A = apply_template(tmpl, params_A);349    std::string output_B = apply_template(tmpl, params_B);350 351    // Check for template application failures352    if (output_A == ERR_TMPL || output_B == ERR_TMPL) {353        return std::nullopt;354    }355 356    // Calculate diff and return result with both outputs357    compare_variants_result result;358    result.diff     = calculate_diff_split(output_A, output_B);359    result.output_A = output_A;360    result.output_B = output_B;361 362    return result;363}364 365}  // namespace autoparser366 367 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai