Felipe97/llama-cpp-compiled
01.2k
1#include "bench.h"2 3#include <algorithm>4#include <chrono>5#include <cmath>6#include <cstdio>7#include <thread>8#include <utility>9 10perf_cell build_perf_cell(ggml_backend_t backend,11 const build_graph_fn & build,12 const init_tensors_fn & init,13 const op_flops_fn & flops) {14 perf_cell cell;15 16 const size_t graph_nodes = 1024;17 18 ggml_init_params params = {19 /* .mem_size = */ ggml_tensor_overhead() * 128 + ggml_graph_overhead_custom(graph_nodes, false),20 /* .mem_base = */ NULL,21 /* .no_alloc = */ true,22 };23 24 cell.ctx.reset(ggml_init(params));25 GGML_ASSERT(cell.ctx);26 27 ggml_tensor * out = build(cell.ctx.get());28 if (!out || !ggml_backend_supports_op(backend, out)) {29 return cell;30 }31 32 cell.buf.reset(ggml_backend_alloc_ctx_tensors(cell.ctx.get(), backend));33 if (!cell.buf) {34 return cell;35 }36 37 init(cell.ctx.get());38 39 cell.gf = ggml_new_graph_custom(cell.ctx.get(), graph_nodes, false);40 ggml_build_forward_expand(cell.gf, out);41 42 // replicate the op to amortize overhead (target ~50 GFLOP/compute, capped to bound graph size)43 cell.n_runs = 1;44 const uint64_t n_flops = flops(out);45 if (n_flops > 0) {46 const uint64_t target_flops = 50ULL * 1000 * 1000 * 1000;47 const int cap = 512;48 const int by_flops = (int) std::min<int64_t>(cap, (int64_t) (target_flops / n_flops));49 cell.n_runs =50 std::max(1, std::min<int>(by_flops, (int) (ggml_graph_size(cell.gf) - ggml_graph_n_nodes(cell.gf))));51 }52 for (int i = 1; i < cell.n_runs; ++i) {53 ggml_graph_add_node(cell.gf, out);54 }55 56 return cell;57}58 59double time_cell_median(ggml_backend_t backend, const perf_cell & cell, int reps) {60 if (cell.gf == nullptr) {61 return -1.0;62 }63 64 ggml_backend_graph_compute(backend, cell.gf); // warmup (compiles the pipeline for this config)65 ggml_backend_synchronize(backend);66 67 std::vector<double> samples;68 samples.reserve(reps);69 for (int r = 0; r < reps; ++r) {70 const int64_t t0 = ggml_time_us();71 ggml_backend_graph_compute(backend, cell.gf);72 ggml_backend_synchronize(backend);73 samples.push_back((double) (ggml_time_us() - t0));74 }75 std::nth_element(samples.begin(), samples.begin() + samples.size() / 2, samples.end());76 77 return samples[samples.size() / 2] / cell.n_runs;78}79 80static double measure_one(ggml_backend_t backend,81 const perf_cell & cell,82 int reps,83 const set_candidate_fn & set_cand,84 const clear_candidate_fn & clear_cand,85 int cand) {86 set_cand(cand);87 const double t = time_cell_median(backend, cell, reps);88 clear_cand();89 90 return t;91}92 93// waits for the anchor to come back within eps of anchor_ref, with exponential backoff.94// returns the converged anchor, or -1 if it never converged within max_wait.95static double cool_until_steady(ggml_backend_t backend,96 const perf_cell & cell,97 int reps,98 const set_candidate_fn & set_cand,99 const clear_candidate_fn & clear_cand,100 int baseline_cand,101 double & anchor_ref,102 const cooldown_opts & cool,103 const char * cell_label) {104 int total_wait = 0;105 106 for (int sleep_s = 2; total_wait < cool.max_wait; sleep_s = std::min(sleep_s * 2, 32)) {107 const int this_wait = std::min(sleep_s, cool.max_wait - total_wait);108 109 fprintf(stderr, "# COOL sleeping %ds (%ds/%ds) %s\n", this_wait, total_wait + this_wait, cool.max_wait,110 cell_label);111 std::this_thread::sleep_for(std::chrono::seconds(this_wait));112 total_wait += this_wait;113 114 const double a = measure_one(backend, cell, reps, set_cand, clear_cand, baseline_cand);115 if (a <= 0.0) {116 continue;117 }118 119 // a faster anchor means the machine got cooler than anything seen so far: adopt it120 if (a < anchor_ref) {121 anchor_ref = a;122 }123 124 if (a <= anchor_ref * (1.0 + cool.eps)) {125 fprintf(stderr, "# COOL steady after %ds %s\n", total_wait, cell_label);126 return a;127 }128 }129 130 fprintf(stderr, "# COOL gave up after %ds %s\n", total_wait, cell_label);131 132 return -1.0;133}134 135cell_result measure_cell(ggml_backend_t backend,136 const perf_cell & cell,137 int reps,138 const std::vector<int> & order,139 const set_candidate_fn & set_cand,140 const clear_candidate_fn & clear_cand,141 int baseline_cand,142 const cooldown_opts & cool,143 const char * cell_label) {144 cell_result res;145 res.t.assign(order.size(), 0.0);146 147 double anchor_ref = 0.0;148 149 // anchors accepted as clean, as (value, position in order[]). the dirty window starts150 // at the position of the last anchor still within eps of anchor_ref, so a downward151 // drift (anchor_ref dropping) naturally widens the window to the whole cell.152 std::vector<std::pair<double, size_t>> anchors;153 154 auto window_start = [&]() -> size_t {155 for (size_t i = anchors.size(); i-- > 0;) {156 if (anchors[i].first <= anchor_ref * (1.0 + cool.eps)) {157 return anchors[i].second;158 }159 }160 return 0; // no clean anchor left -> the whole cell is suspect161 };162 163 int retries_left = cool.max_retry;164 165 for (size_t i = 0; i < order.size(); ++i) {166 res.t[order[i]] = measure_one(backend, cell, reps, set_cand, clear_cand, order[i]);167 168 if (i % 4 != 0) {169 continue;170 }171 172 const double a = measure_one(backend, cell, reps, set_cand, clear_cand, baseline_cand);173 if (a <= 0.0) {174 continue;175 }176 177 res.anchor_min = res.anchor_min > 0.0 ? std::min(res.anchor_min, a) : a;178 res.anchor_max = std::max(res.anchor_max, a);179 180 if (anchor_ref == 0.0) {181 anchor_ref = a;182 anchors.push_back({ a, i });183 continue;184 }185 186 const double drift = std::fabs(a - anchor_ref) / anchor_ref;187 188 // a cooler anchor than any so far becomes the reference: whatever was measured189 // before it was measured on a hotter machine190 if (a < anchor_ref) {191 anchor_ref = a;192 }193 194 if (drift <= cool.drift) {195 anchors.push_back({ a, i });196 continue;197 }198 199 fprintf(stderr, "# WARN throttling? anchor drift %.1f%% %s\n", 100.0 * drift, cell_label);200 201 if (!cool.enabled) {202 anchors.push_back({ a, i });203 continue;204 }205 206 if (retries_left <= 0) {207 fprintf(stderr, "# DIRTY retries exhausted %s\n", cell_label);208 res.trusted = false;209 return res;210 }211 212 const size_t dirty_from = window_start();213 214 const double a_cool =215 cool_until_steady(backend, cell, reps, set_cand, clear_cand, baseline_cand, anchor_ref, cool, cell_label);216 if (a_cool <= 0.0) {217 res.trusted = false;218 return res;219 }220 221 // the converged anchor is the only clean one now; re-measure the dirty window from it222 anchors.clear();223 anchors.push_back({ a_cool, dirty_from });224 225 retries_left--;226 227 fprintf(stderr, "# REDO candidates %zu..%zu %s\n", dirty_from, i, cell_label);228 for (size_t j = dirty_from; j <= i; ++j) {229 res.t[order[j]] = measure_one(backend, cell, reps, set_cand, clear_cand, order[j]);230 }231 }232 233 return res;234}235 