Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
bench.cpp235 linesDownload Raw Back to tuning
1#include "bench.h"2 3#include <algorithm>4#include <chrono>5#include <cmath>6#include <cstdio>7#include <thread>8#include <utility>9 10perf_cell build_perf_cell(ggml_backend_t          backend,11                          const build_graph_fn &  build,12                          const init_tensors_fn & init,13                          const op_flops_fn &     flops) {14    perf_cell cell;15 16    const size_t graph_nodes = 1024;17 18    ggml_init_params params = {19        /* .mem_size  = */ ggml_tensor_overhead() * 128 + ggml_graph_overhead_custom(graph_nodes, false),20        /* .mem_base  = */ NULL,21        /* .no_alloc  = */ true,22    };23 24    cell.ctx.reset(ggml_init(params));25    GGML_ASSERT(cell.ctx);26 27    ggml_tensor * out = build(cell.ctx.get());28    if (!out || !ggml_backend_supports_op(backend, out)) {29        return cell;30    }31 32    cell.buf.reset(ggml_backend_alloc_ctx_tensors(cell.ctx.get(), backend));33    if (!cell.buf) {34        return cell;35    }36 37    init(cell.ctx.get());38 39    cell.gf = ggml_new_graph_custom(cell.ctx.get(), graph_nodes, false);40    ggml_build_forward_expand(cell.gf, out);41 42    // replicate the op to amortize overhead (target ~50 GFLOP/compute, capped to bound graph size)43    cell.n_runs            = 1;44    const uint64_t n_flops = flops(out);45    if (n_flops > 0) {46        const uint64_t target_flops = 50ULL * 1000 * 1000 * 1000;47        const int      cap          = 512;48        const int      by_flops     = (int) std::min<int64_t>(cap, (int64_t) (target_flops / n_flops));49        cell.n_runs =50            std::max(1, std::min<int>(by_flops, (int) (ggml_graph_size(cell.gf) - ggml_graph_n_nodes(cell.gf))));51    }52    for (int i = 1; i < cell.n_runs; ++i) {53        ggml_graph_add_node(cell.gf, out);54    }55 56    return cell;57}58 59double time_cell_median(ggml_backend_t backend, const perf_cell & cell, int reps) {60    if (cell.gf == nullptr) {61        return -1.0;62    }63 64    ggml_backend_graph_compute(backend, cell.gf);  // warmup (compiles the pipeline for this config)65    ggml_backend_synchronize(backend);66 67    std::vector<double> samples;68    samples.reserve(reps);69    for (int r = 0; r < reps; ++r) {70        const int64_t t0 = ggml_time_us();71        ggml_backend_graph_compute(backend, cell.gf);72        ggml_backend_synchronize(backend);73        samples.push_back((double) (ggml_time_us() - t0));74    }75    std::nth_element(samples.begin(), samples.begin() + samples.size() / 2, samples.end());76 77    return samples[samples.size() / 2] / cell.n_runs;78}79 80static double measure_one(ggml_backend_t             backend,81                          const perf_cell &          cell,82                          int                        reps,83                          const set_candidate_fn &   set_cand,84                          const clear_candidate_fn & clear_cand,85                          int                        cand) {86    set_cand(cand);87    const double t = time_cell_median(backend, cell, reps);88    clear_cand();89 90    return t;91}92 93// waits for the anchor to come back within eps of anchor_ref, with exponential backoff.94// returns the converged anchor, or -1 if it never converged within max_wait.95static double cool_until_steady(ggml_backend_t             backend,96                                const perf_cell &          cell,97                                int                        reps,98                                const set_candidate_fn &   set_cand,99                                const clear_candidate_fn & clear_cand,100                                int                        baseline_cand,101                                double &                   anchor_ref,102                                const cooldown_opts &      cool,103                                const char *               cell_label) {104    int total_wait = 0;105 106    for (int sleep_s = 2; total_wait < cool.max_wait; sleep_s = std::min(sleep_s * 2, 32)) {107        const int this_wait = std::min(sleep_s, cool.max_wait - total_wait);108 109        fprintf(stderr, "# COOL sleeping %ds (%ds/%ds) %s\n", this_wait, total_wait + this_wait, cool.max_wait,110                cell_label);111        std::this_thread::sleep_for(std::chrono::seconds(this_wait));112        total_wait += this_wait;113 114        const double a = measure_one(backend, cell, reps, set_cand, clear_cand, baseline_cand);115        if (a <= 0.0) {116            continue;117        }118 119        // a faster anchor means the machine got cooler than anything seen so far: adopt it120        if (a < anchor_ref) {121            anchor_ref = a;122        }123 124        if (a <= anchor_ref * (1.0 + cool.eps)) {125            fprintf(stderr, "# COOL steady after %ds %s\n", total_wait, cell_label);126            return a;127        }128    }129 130    fprintf(stderr, "# COOL gave up after %ds %s\n", total_wait, cell_label);131 132    return -1.0;133}134 135cell_result measure_cell(ggml_backend_t             backend,136                         const perf_cell &          cell,137                         int                        reps,138                         const std::vector<int> &   order,139                         const set_candidate_fn &   set_cand,140                         const clear_candidate_fn & clear_cand,141                         int                        baseline_cand,142                         const cooldown_opts &      cool,143                         const char *               cell_label) {144    cell_result res;145    res.t.assign(order.size(), 0.0);146 147    double anchor_ref = 0.0;148 149    // anchors accepted as clean, as (value, position in order[]). the dirty window starts150    // at the position of the last anchor still within eps of anchor_ref, so a downward151    // drift (anchor_ref dropping) naturally widens the window to the whole cell.152    std::vector<std::pair<double, size_t>> anchors;153 154    auto window_start = [&]() -> size_t {155        for (size_t i = anchors.size(); i-- > 0;) {156            if (anchors[i].first <= anchor_ref * (1.0 + cool.eps)) {157                return anchors[i].second;158            }159        }160        return 0;  // no clean anchor left -> the whole cell is suspect161    };162 163    int retries_left = cool.max_retry;164 165    for (size_t i = 0; i < order.size(); ++i) {166        res.t[order[i]] = measure_one(backend, cell, reps, set_cand, clear_cand, order[i]);167 168        if (i % 4 != 0) {169            continue;170        }171 172        const double a = measure_one(backend, cell, reps, set_cand, clear_cand, baseline_cand);173        if (a <= 0.0) {174            continue;175        }176 177        res.anchor_min = res.anchor_min > 0.0 ? std::min(res.anchor_min, a) : a;178        res.anchor_max = std::max(res.anchor_max, a);179 180        if (anchor_ref == 0.0) {181            anchor_ref = a;182            anchors.push_back({ a, i });183            continue;184        }185 186        const double drift = std::fabs(a - anchor_ref) / anchor_ref;187 188        // a cooler anchor than any so far becomes the reference: whatever was measured189        // before it was measured on a hotter machine190        if (a < anchor_ref) {191            anchor_ref = a;192        }193 194        if (drift <= cool.drift) {195            anchors.push_back({ a, i });196            continue;197        }198 199        fprintf(stderr, "# WARN throttling? anchor drift %.1f%% %s\n", 100.0 * drift, cell_label);200 201        if (!cool.enabled) {202            anchors.push_back({ a, i });203            continue;204        }205 206        if (retries_left <= 0) {207            fprintf(stderr, "# DIRTY retries exhausted %s\n", cell_label);208            res.trusted = false;209            return res;210        }211 212        const size_t dirty_from = window_start();213 214        const double a_cool =215            cool_until_steady(backend, cell, reps, set_cand, clear_cand, baseline_cand, anchor_ref, cool, cell_label);216        if (a_cool <= 0.0) {217            res.trusted = false;218            return res;219        }220 221        // the converged anchor is the only clean one now; re-measure the dirty window from it222        anchors.clear();223        anchors.push_back({ a_cool, dirty_from });224 225        retries_left--;226 227        fprintf(stderr, "# REDO candidates %zu..%zu %s\n", dirty_from, i, cell_label);228        for (size_t j = dirty_from; j <= i; ++j) {229            res.t[order[j]] = measure_one(backend, cell, reps, set_cand, clear_cand, order[j]);230        }231    }232 233    return res;234}235