KBaba7/llama.cpp
0
1# TODO: there have been some issues with the workflow, so disabling for now2# https://github.com/ggerganov/llama.cpp/issues/78933#4# Benchmark5name: Benchmark6 7on:8 workflow_dispatch:9 inputs:10 gpu-series:11 description: 'Azure GPU series to run with'12 required: true13 type: choice14 options:15 - Standard_NC4as_T4_v316 - Standard_NC24ads_A100_v417 - Standard_NC80adis_H100_v518 sha:19 description: 'Commit SHA1 to build'20 required: false21 type: string22 duration:23 description: 'Duration of the bench'24 type: string25 default: 10m26 27 push:28 branches:29 - master30 paths: ['llama.cpp', 'ggml.c', 'ggml-backend.cpp', 'ggml-quants.c', '**/*.cu', 'examples/server/*.h*', 'examples/server/*.cpp']31 pull_request_target:32 types: [opened, synchronize, reopened]33 paths: ['llama.cpp', 'ggml.c', 'ggml-backend.cpp', 'ggml-quants.c', '**/*.cu', 'examples/server/*.h*', 'examples/server/*.cpp']34 schedule:35 - cron: '04 2 * * *'36 37concurrency:38 group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}-${{ github.event.inputs.sha }}39 cancel-in-progress: true40 41jobs:42 bench-server-baseline:43 runs-on: Standard_NC4as_T4_v344 env:45 RUNNER_LABEL: Standard_NC4as_T4_v3 # FIXME Do not find a way to not duplicate it46 N_USERS: 847 DURATION: 10m48 49 strategy:50 matrix:51 model: [phi-2]52 ftype: [q4_0, q8_0, f16]53 include:54 - model: phi-255 ftype: q4_056 pr_comment_enabled: "true"57 58 if: |59 inputs.gpu-series == 'Standard_NC4as_T4_v3'60 || (61 github.event_name == 'schedule'62 && github.ref_name == 'master'63 && github.repository_owner == 'ggerganov'64 )65 || github.event_name == 'pull_request_target'66 || (67 github.event_name == 'push'68 && github.event.ref == 'refs/heads/master'69 && github.repository_owner == 'ggerganov'70 )71 steps:72 - name: Clone73 id: checkout74 uses: actions/checkout@v475 with:76 fetch-depth: 077 ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}78 79 - name: Install python env80 id: pipenv81 run: |82 cd examples/server/bench83 python3 -m venv venv84 source venv/bin/activate85 pip install -r requirements.txt86 87 - name: Prometheus88 id: install_prometheus89 run: |90 wget --quiet https://github.com/prometheus/prometheus/releases/download/v2.51.0/prometheus-2.51.0.linux-amd64.tar.gz91 tar xzf prometheus*.tar.gz --strip-components=192 ./prometheus --config.file=examples/server/bench/prometheus.yml &93 while ! nc -z localhost 9090; do94 sleep 0.195 done96 97 - name: Set up Go98 uses: actions/setup-go@v599 with:100 go-version: '1.21'101 102 - name: Install k6 and xk6-sse103 id: k6_installation104 run: |105 cd examples/server/bench106 go install go.k6.io/xk6/cmd/xk6@latest107 xk6 build master \108 --with github.com/phymbert/xk6-sse109 110 - name: Build111 id: cmake_build112 run: |113 set -eux114 cmake -B build \115 -DGGML_NATIVE=OFF \116 -DLLAMA_BUILD_SERVER=ON \117 -DLLAMA_CURL=ON \118 -DLLAMA_CUBLAS=ON \119 -DCUDAToolkit_ROOT=/usr/local/cuda \120 -DCMAKE_CUDA_COMPILER=/usr/local/cuda/bin/nvcc \121 -DCMAKE_CUDA_ARCHITECTURES=75 \122 -DLLAMA_FATAL_WARNINGS=OFF \123 -DLLAMA_ALL_WARNINGS=OFF \124 -DCMAKE_BUILD_TYPE=Release;125 cmake --build build --config Release -j $(nproc) --target llama-server126 127 - name: Download the dataset128 id: download_dataset129 run: |130 cd examples/server/bench131 wget --quiet https://huggingface.co/datasets/anon8231489123/ShareGPT_Vicuna_unfiltered/resolve/main/ShareGPT_V3_unfiltered_cleaned_split.json132 133 - name: Server bench134 id: server_bench135 env:136 HEAD_REF: ${{ github.head_ref || github.ref_name }}137 run: |138 set -eux139 140 cd examples/server/bench141 source venv/bin/activate142 python bench.py \143 --runner-label ${{ env.RUNNER_LABEL }} \144 --name ${{ github.job }} \145 --branch $HEAD_REF \146 --commit ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha }} \147 --scenario script.js \148 --duration ${{ github.event.inputs.duration || env.DURATION }} \149 --hf-repo ggml-org/models \150 --hf-file ${{ matrix.model }}/ggml-model-${{ matrix.ftype }}.gguf \151 --model-path-prefix /models \152 --parallel ${{ env.N_USERS }} \153 -ngl 33 \154 --batch-size 2048 \155 --ubatch-size 256 \156 --ctx-size 16384 \157 --n-prompts 1000 \158 --max-prompt-tokens 1024 \159 --max-tokens 2048160 161 cat results.github.env >> $GITHUB_ENV162 163 # Remove dataset as we do not want it in the artefact164 rm ShareGPT_V3_unfiltered_cleaned_split.json165 166 - uses: actions/upload-artifact@v4167 with:168 name: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}169 compression-level: 9170 path: |171 examples/server/bench/*.jpg172 examples/server/bench/*.json173 examples/server/bench/*.log174 175 - name: Commit status176 uses: Sibz/github-status-action@v1177 with:178 authToken: ${{secrets.GITHUB_TOKEN}}179 sha: ${{ inputs.sha || github.event.pull_request.head.sha || github.sha }}180 context: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}181 description: |182 ${{ env.BENCH_RESULTS }}183 state: 'success'184 185 - name: Upload benchmark images186 uses: devicons/public-upload-to-imgur@v2.2.2187 continue-on-error: true # Important as it looks unstable: 503188 id: imgur_step189 with:190 client_id: ${{secrets.IMGUR_CLIENT_ID}}191 path: |192 examples/server/bench/prompt_tokens_seconds.jpg193 examples/server/bench/predicted_tokens_seconds.jpg194 examples/server/bench/kv_cache_usage_ratio.jpg195 examples/server/bench/requests_processing.jpg196 197 - name: Extract mermaid198 id: set_mermaid199 run: |200 set -eux201 202 cd examples/server/bench203 PROMPT_TOKENS_SECONDS=$(cat prompt_tokens_seconds.mermaid)204 echo "PROMPT_TOKENS_SECONDS<<EOF" >> $GITHUB_ENV205 echo "$PROMPT_TOKENS_SECONDS" >> $GITHUB_ENV206 echo "EOF" >> $GITHUB_ENV207 208 PREDICTED_TOKENS_SECONDS=$(cat predicted_tokens_seconds.mermaid)209 echo "PREDICTED_TOKENS_SECONDS<<EOF" >> $GITHUB_ENV210 echo "$PREDICTED_TOKENS_SECONDS" >> $GITHUB_ENV211 echo "EOF" >> $GITHUB_ENV212 213 KV_CACHE_USAGE_RATIO=$(cat kv_cache_usage_ratio.mermaid)214 echo "KV_CACHE_USAGE_RATIO<<EOF" >> $GITHUB_ENV215 echo "$KV_CACHE_USAGE_RATIO" >> $GITHUB_ENV216 echo "EOF" >> $GITHUB_ENV217 218 REQUESTS_PROCESSING=$(cat requests_processing.mermaid)219 echo "REQUESTS_PROCESSING<<EOF" >> $GITHUB_ENV220 echo "$REQUESTS_PROCESSING" >> $GITHUB_ENV221 echo "EOF" >> $GITHUB_ENV222 223 - name: Extract image url224 id: extract_image_url225 continue-on-error: true226 run: |227 set -eux228 229 echo "IMAGE_O=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[0] }}" >> $GITHUB_ENV230 echo "IMAGE_1=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[1] }}" >> $GITHUB_ENV231 echo "IMAGE_2=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[2] }}" >> $GITHUB_ENV232 echo "IMAGE_3=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[3] }}" >> $GITHUB_ENV233 234 - name: Comment PR235 uses: mshick/add-pr-comment@v2236 id: comment_pr237 if: ${{ github.event.pull_request != '' && matrix.pr_comment_enabled == 'true' }}238 with:239 message-id: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}240 message: |241 <p align="center">242 243 ๐ **llama.cpp server** for _${{ github.job }}_ on _${{ env.RUNNER_LABEL }}_ for `${{ matrix.model }}`-`${{ matrix.ftype }}`: **${{ env.BENCH_ITERATIONS}} iterations** ๐244 245 </p>246 247 <details>248 249 <summary>Expand details for performance related PR only</summary>250 251 - Concurrent users: ${{ env.N_USERS }}, duration: ${{ github.event.inputs.duration || env.DURATION }}252 - HTTP request : avg=${{ env.HTTP_REQ_DURATION_AVG }}ms p(95)=${{ env.HTTP_REQ_DURATION_P_95_ }}ms fails=${{ env.HTTP_REQ_FAILED_PASSES }}, finish reason: stop=${{ env.LLAMACPP_COMPLETIONS_STOP_RATE_PASSES }} truncated=${{ env.LLAMACPP_COMPLETIONS_TRUNCATED_RATE_PASSES }}253 - Prompt processing (pp): avg=${{ env.LLAMACPP_PROMPT_PROCESSING_SECOND_AVG }}tk/s p(95)=${{ env.LLAMACPP_PROMPT_PROCESSING_SECOND_P_95_ }}tk/s254 - Token generation (tg): avg=${{ env.LLAMACPP_TOKENS_SECOND_AVG }}tk/s p(95)=${{ env.LLAMACPP_TOKENS_SECOND_P_95_ }}tk/s255 - ${{ env.BENCH_GRAPH_XLABEL }}256 257 258 <p align="center">259 260 <img width="100%" height="100%" src="${{ env.IMAGE_O }}" alt="prompt_tokens_seconds" />261 262 <details>263 264 <summary>More</summary>265 266 ```mermaid267 ${{ env.PROMPT_TOKENS_SECONDS }}268 ```269 270 </details>271 272 <img width="100%" height="100%" src="${{ env.IMAGE_1 }}" alt="predicted_tokens_seconds"/>273 274 <details>275 <summary>More</summary>276 277 ```mermaid278 ${{ env.PREDICTED_TOKENS_SECONDS }}279 ```280 281 </details>282 283 </p>284 285 <details>286 287 <summary>Details</summary>288 289 <p align="center">290 291 <img width="100%" height="100%" src="${{ env.IMAGE_2 }}" alt="kv_cache_usage_ratio" />292 293 <details>294 <summary>More</summary>295 296 ```mermaid297 ${{ env.KV_CACHE_USAGE_RATIO }}298 ```299 300 </details>301 302 <img width="100%" height="100%" src="${{ env.IMAGE_3 }}" alt="requests_processing"/>303 304 <details>305 <summary>More</summary>306 307 ```mermaid308 ${{ env.REQUESTS_PROCESSING }}309 ```310 311 </details>312 313 </p>314 </details>315 </details>316 