Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
bench.yml.disabled316 linesDownload Raw Back to workflows
1# TODO: there have been some issues with the workflow, so disabling for now2#       https://github.com/ggerganov/llama.cpp/issues/78933#4# Benchmark5name: Benchmark6 7on:8  workflow_dispatch:9    inputs:10      gpu-series:11        description: 'Azure GPU series to run with'12        required: true13        type: choice14        options:15          - Standard_NC4as_T4_v316          - Standard_NC24ads_A100_v417          - Standard_NC80adis_H100_v518      sha:19        description: 'Commit SHA1 to build'20        required: false21        type: string22      duration:23        description: 'Duration of the bench'24        type: string25        default: 10m26 27  push:28    branches:29      - master30    paths: ['llama.cpp', 'ggml.c', 'ggml-backend.cpp', 'ggml-quants.c', '**/*.cu', 'examples/server/*.h*', 'examples/server/*.cpp']31  pull_request_target:32    types: [opened, synchronize, reopened]33    paths: ['llama.cpp', 'ggml.c', 'ggml-backend.cpp', 'ggml-quants.c', '**/*.cu', 'examples/server/*.h*', 'examples/server/*.cpp']34  schedule:35    -  cron: '04 2 * * *'36 37concurrency:38  group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}-${{ github.event.inputs.sha }}39  cancel-in-progress: true40 41jobs:42  bench-server-baseline:43    runs-on: Standard_NC4as_T4_v344    env:45      RUNNER_LABEL: Standard_NC4as_T4_v3 # FIXME Do not find a way to not duplicate it46      N_USERS: 847      DURATION: 10m48 49    strategy:50      matrix:51        model: [phi-2]52        ftype: [q4_0, q8_0, f16]53        include:54          - model: phi-255            ftype: q4_056            pr_comment_enabled: "true"57 58    if: |59      inputs.gpu-series == 'Standard_NC4as_T4_v3'60      || (61        github.event_name == 'schedule'62        && github.ref_name == 'master'63        && github.repository_owner == 'ggerganov'64      )65      || github.event_name == 'pull_request_target'66      || (67        github.event_name == 'push'68        && github.event.ref == 'refs/heads/master'69        && github.repository_owner == 'ggerganov'70      )71    steps:72      - name: Clone73        id: checkout74        uses: actions/checkout@v475        with:76          fetch-depth: 077          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}78 79      - name: Install python env80        id: pipenv81        run: |82          cd examples/server/bench83          python3 -m venv venv84          source venv/bin/activate85          pip install -r requirements.txt86 87      - name: Prometheus88        id: install_prometheus89        run: |90          wget --quiet https://github.com/prometheus/prometheus/releases/download/v2.51.0/prometheus-2.51.0.linux-amd64.tar.gz91          tar xzf prometheus*.tar.gz --strip-components=192          ./prometheus --config.file=examples/server/bench/prometheus.yml &93          while ! nc -z localhost 9090; do94            sleep 0.195          done96 97      - name: Set up Go98        uses: actions/setup-go@v599        with:100          go-version: '1.21'101 102      - name: Install k6 and xk6-sse103        id: k6_installation104        run: |105          cd examples/server/bench106          go install go.k6.io/xk6/cmd/xk6@latest107          xk6 build master \108              --with github.com/phymbert/xk6-sse109 110      - name: Build111        id: cmake_build112        run: |113          set -eux114          cmake -B build \115              -DGGML_NATIVE=OFF \116              -DLLAMA_BUILD_SERVER=ON \117              -DLLAMA_CURL=ON \118              -DLLAMA_CUBLAS=ON \119              -DCUDAToolkit_ROOT=/usr/local/cuda \120              -DCMAKE_CUDA_COMPILER=/usr/local/cuda/bin/nvcc \121              -DCMAKE_CUDA_ARCHITECTURES=75 \122              -DLLAMA_FATAL_WARNINGS=OFF \123              -DLLAMA_ALL_WARNINGS=OFF \124              -DCMAKE_BUILD_TYPE=Release;125          cmake --build build --config Release -j $(nproc) --target llama-server126 127      - name: Download the dataset128        id: download_dataset129        run: |130          cd examples/server/bench131          wget --quiet https://huggingface.co/datasets/anon8231489123/ShareGPT_Vicuna_unfiltered/resolve/main/ShareGPT_V3_unfiltered_cleaned_split.json132 133      - name: Server bench134        id: server_bench135        env:136            HEAD_REF: ${{ github.head_ref || github.ref_name }}137        run: |138          set -eux139 140          cd examples/server/bench141          source venv/bin/activate142          python bench.py \143              --runner-label ${{ env.RUNNER_LABEL }} \144              --name ${{ github.job }} \145              --branch $HEAD_REF \146              --commit ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha }} \147              --scenario script.js \148              --duration ${{ github.event.inputs.duration || env.DURATION }} \149              --hf-repo ggml-org/models	 \150              --hf-file ${{ matrix.model }}/ggml-model-${{ matrix.ftype }}.gguf \151              --model-path-prefix /models \152              --parallel ${{ env.N_USERS }} \153              -ngl 33 \154              --batch-size 2048 \155              --ubatch-size	256 \156              --ctx-size 16384 \157              --n-prompts 1000 \158              --max-prompt-tokens 1024 \159              --max-tokens 2048160 161          cat results.github.env >> $GITHUB_ENV162 163          # Remove dataset as we do not want it in the artefact164          rm ShareGPT_V3_unfiltered_cleaned_split.json165 166      - uses: actions/upload-artifact@v4167        with:168          name: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}169          compression-level: 9170          path: |171            examples/server/bench/*.jpg172            examples/server/bench/*.json173            examples/server/bench/*.log174 175      - name: Commit status176        uses: Sibz/github-status-action@v1177        with:178          authToken: ${{secrets.GITHUB_TOKEN}}179          sha: ${{ inputs.sha || github.event.pull_request.head.sha || github.sha }}180          context: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}181          description: |182            ${{ env.BENCH_RESULTS }}183          state: 'success'184 185      - name: Upload benchmark images186        uses: devicons/public-upload-to-imgur@v2.2.2187        continue-on-error: true # Important as it looks unstable: 503188        id: imgur_step189        with:190          client_id: ${{secrets.IMGUR_CLIENT_ID}}191          path: |192            examples/server/bench/prompt_tokens_seconds.jpg193            examples/server/bench/predicted_tokens_seconds.jpg194            examples/server/bench/kv_cache_usage_ratio.jpg195            examples/server/bench/requests_processing.jpg196 197      - name: Extract mermaid198        id: set_mermaid199        run: |200          set -eux201 202          cd examples/server/bench203          PROMPT_TOKENS_SECONDS=$(cat prompt_tokens_seconds.mermaid)204          echo "PROMPT_TOKENS_SECONDS<<EOF" >> $GITHUB_ENV205          echo "$PROMPT_TOKENS_SECONDS" >> $GITHUB_ENV206          echo "EOF" >> $GITHUB_ENV207 208          PREDICTED_TOKENS_SECONDS=$(cat predicted_tokens_seconds.mermaid)209          echo "PREDICTED_TOKENS_SECONDS<<EOF" >> $GITHUB_ENV210          echo "$PREDICTED_TOKENS_SECONDS" >> $GITHUB_ENV211          echo "EOF" >> $GITHUB_ENV212 213          KV_CACHE_USAGE_RATIO=$(cat kv_cache_usage_ratio.mermaid)214          echo "KV_CACHE_USAGE_RATIO<<EOF" >> $GITHUB_ENV215          echo "$KV_CACHE_USAGE_RATIO" >> $GITHUB_ENV216          echo "EOF" >> $GITHUB_ENV217 218          REQUESTS_PROCESSING=$(cat requests_processing.mermaid)219          echo "REQUESTS_PROCESSING<<EOF" >> $GITHUB_ENV220          echo "$REQUESTS_PROCESSING" >> $GITHUB_ENV221          echo "EOF" >> $GITHUB_ENV222 223      - name: Extract image url224        id: extract_image_url225        continue-on-error: true226        run: |227          set -eux228 229          echo "IMAGE_O=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[0] }}" >> $GITHUB_ENV230          echo "IMAGE_1=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[1] }}" >> $GITHUB_ENV231          echo "IMAGE_2=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[2] }}" >> $GITHUB_ENV232          echo "IMAGE_3=${{ fromJSON(steps.imgur_step.outputs.imgur_urls)[3] }}" >> $GITHUB_ENV233 234      - name: Comment PR235        uses: mshick/add-pr-comment@v2236        id: comment_pr237        if: ${{ github.event.pull_request != '' && matrix.pr_comment_enabled == 'true' }}238        with:239          message-id: bench-server-${{ github.job }}-${{ env.RUNNER_LABEL }}-${{ matrix.model }}-${{ matrix.ftype }}240          message: |241            <p align="center">242 243            ๐Ÿ“ˆ **llama.cpp server** for _${{ github.job }}_ on _${{ env.RUNNER_LABEL }}_ for `${{ matrix.model }}`-`${{ matrix.ftype }}`: **${{ env.BENCH_ITERATIONS}} iterations** ๐Ÿš€244 245            </p>246 247            <details>248 249            <summary>Expand details for performance related PR only</summary>250 251            - Concurrent users: ${{ env.N_USERS }}, duration: ${{ github.event.inputs.duration || env.DURATION }}252            - HTTP request          : avg=${{ env.HTTP_REQ_DURATION_AVG }}ms        p(95)=${{ env.HTTP_REQ_DURATION_P_95_ }}ms fails=${{ env.HTTP_REQ_FAILED_PASSES }}, finish reason: stop=${{ env.LLAMACPP_COMPLETIONS_STOP_RATE_PASSES }} truncated=${{ env.LLAMACPP_COMPLETIONS_TRUNCATED_RATE_PASSES }}253            - Prompt processing (pp): avg=${{ env.LLAMACPP_PROMPT_PROCESSING_SECOND_AVG }}tk/s p(95)=${{ env.LLAMACPP_PROMPT_PROCESSING_SECOND_P_95_ }}tk/s254            - Token generation  (tg): avg=${{ env.LLAMACPP_TOKENS_SECOND_AVG }}tk/s p(95)=${{ env.LLAMACPP_TOKENS_SECOND_P_95_ }}tk/s255            - ${{ env.BENCH_GRAPH_XLABEL }}256 257 258            <p align="center">259 260            <img width="100%" height="100%" src="${{ env.IMAGE_O }}" alt="prompt_tokens_seconds" />261 262            <details>263 264            <summary>More</summary>265 266            ```mermaid267            ${{ env.PROMPT_TOKENS_SECONDS }}268            ```269 270            </details>271 272            <img width="100%" height="100%" src="${{ env.IMAGE_1 }}" alt="predicted_tokens_seconds"/>273 274            <details>275                <summary>More</summary>276 277            ```mermaid278            ${{ env.PREDICTED_TOKENS_SECONDS }}279            ```280 281            </details>282 283            </p>284 285            <details>286 287            <summary>Details</summary>288 289            <p align="center">290 291            <img width="100%" height="100%" src="${{ env.IMAGE_2 }}" alt="kv_cache_usage_ratio" />292 293            <details>294                <summary>More</summary>295 296            ```mermaid297            ${{ env.KV_CACHE_USAGE_RATIO }}298            ```299 300            </details>301 302            <img width="100%" height="100%" src="${{ env.IMAGE_3 }}" alt="requests_processing"/>303 304            <details>305                <summary>More</summary>306 307            ```mermaid308            ${{ env.REQUESTS_PROCESSING }}309            ```310 311            </details>312 313            </p>314            </details>315            </details>316