Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
build-self-hosted.yml540 linesDownload Raw Back to workflows
1name: CI (self-hosted)2 3on:4  workflow_dispatch: # allows manual triggering5  push:6    branches:7      - master8    paths: [9      '.github/workflows/build-self-hosted.yml',10      'ci/run.sh',11      '**/CMakeLists.txt',12      '**/.cmake',13      '**/*.h',14      '**/*.hpp',15      '**/*.c',16      '**/*.cpp',17      '**/*.cu',18      '**/*.cuh',19      '**/*.swift',20      '**/*.m',21      '**/*.metal',22      '**/*.comp',23      '**/*.glsl',24      '**/*.wgsl'25    ]26 27  pull_request:28    types: [opened, synchronize, reopened]29    paths: [30      '.github/workflows/build-self-hosted.yml',31      'ci/run.sh',32      '**/CMakeLists.txt',33      '**/.cmake',34      '**/*.h',35      '**/*.hpp',36      '**/*.c',37      '**/*.cpp',38      '**/*.cu',39      '**/*.cuh',40      '**/*.swift',41      '**/*.m',42      '**/*.metal',43      '**/*.comp',44      '**/*.glsl',45      '**/*.wgsl'46    ]47 48concurrency:49  group: ${{ github.workflow }}-${{ github.head_ref && github.ref || github.run_id }}50  cancel-in-progress: true51 52env:53  # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)54  HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}55  GGML_NLOOP: 356  GGML_N_THREADS: 157  LLAMA_ARG_LOG_COLORS: 158  LLAMA_ARG_LOG_PREFIX: 159  LLAMA_ARG_LOG_TIMESTAMPS: 160 61jobs:62  gpu-cuda:63    runs-on: "hf-jobs-t4-small:cuda13"64 65    steps:66      - name: Clone67        id: checkout68        uses: actions/checkout@v669 70      - name: Install dependencies71        run: |72          sudo apt update73          sudo apt install -y cmake libssl-dev time unzip wget python3 python3-venv python3-pip74 75      - name: ccache76        uses: ggml-org/ccache-action@v1.2.2477        with:78          restore: false79          save: false80 81      - name: ccache-buckets-restore82        uses: ./.github/actions/ccache-buckets83        with:84          key: self-hosted-gpu-cuda85          folder: llama.cpp86          hf_bucket: ggml-org/cache87 88      - name: Test89        id: ggml-ci90        run: |91          nvidia-smi92          GG_BUILD_CUDA=1 CUDACXX=/usr/local/cuda/bin/nvcc bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp93 94      - name: ccache-buckets-save95        if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}96        uses: ./.github/actions/ccache-buckets97        env:98          HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}99        with:100          key: self-hosted-gpu-cuda101          folder: llama.cpp102          evict-old-files: 1d103          hf_bucket: ggml-org/cache104          save: true105 106  gpu-rocm:107    runs-on: [self-hosted, Linux, AMD]108 109    steps:110      - name: Clone111        id: checkout112        uses: actions/checkout@v6113 114      - name: Test115        id: ggml-ci116        # HIP_LAUNCH_BLOCKING=1: workaround for an async-execution correctness117        # issue on integrated RDNA3.5 (gfx1151) where batched inference returns118        # incorrect output (perplexity ~88 vs ~9.4). Serializing kernel launches119        # restores correctness. Remove once the underlying ROCm/HIP issue is fixed.120        env:121          HIP_LAUNCH_BLOCKING: "1"122        run: |123          rocminfo124          GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp125 126  gpu-vulkan-nvidia-cm:127    # runs-on: "hf-jobs-t4-small:ubuntu26_04"128    runs-on: [self-hosted, Linux, NVIDIA]129 130    steps:131      - name: Clone132        id: checkout133        uses: actions/checkout@v6134 135      # - name: Install dependencies136      #   run: |137      #     sudo apt update138      #     sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip139 140      # - name: ccache141      #   uses: ggml-org/ccache-action@v1.2.24142      #   with:143      #     restore: false144      #     save: false145 146      # - name: ccache-buckets-restore147      #   uses: ./.github/actions/ccache-buckets148      #   with:149      #     key: self-hosted-vulkan-nvidia-cm150      #     folder: llama.cpp151      #     hf_bucket: ggml-org/cache152 153      - name: Test154        id: ggml-ci155        run: |156          vulkaninfo --summary157          GG_BUILD_VULKAN=1 GGML_VK_DISABLE_COOPMAT2=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp158 159      # - name: ccache-buckets-save160      #   if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}161      #   uses: ./.github/actions/ccache-buckets162      #   env:163      #     HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}164      #   with:165      #     key: self-hosted-vulkan-nvidia-cm166      #     folder: llama.cpp167      #     evict-old-files: 1d168      #     hf_bucket: ggml-org/cache169      #     save: true170 171  gpu-vulkan-nvidia-cm2:172    # runs-on: "hf-jobs-t4-small:ubuntu26_04"173    runs-on: [self-hosted, Linux, NVIDIA, COOPMAT2]174 175    steps:176      - name: Clone177        id: checkout178        uses: actions/checkout@v6179 180      # - name: Install dependencies181      #   run: |182      #     sudo apt update183      #     sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan-dev glslc spirv-headers vulkan-tools mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip184 185      # - name: ccache186      #   uses: ggml-org/ccache-action@v1.2.24187      #   with:188      #     restore: false189      #     save: false190 191      # - name: ccache-buckets-restore192      #   uses: ./.github/actions/ccache-buckets193      #   with:194      #     key: self-hosted-vulkan-nvidia-cm2195      #     folder: llama.cpp196      #     hf_bucket: ggml-org/cache197 198      - name: Test199        id: ggml-ci200        run: |201          vulkaninfo --summary202          GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp203 204      # - name: ccache-buckets-save205      #   if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}206      #   uses: ./.github/actions/ccache-buckets207      #   env:208      #     HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}209      #   with:210      #     key: self-hosted-vulkan-nvidia-cm2211      #     folder: llama.cpp212      #     evict-old-files: 1d213      #     hf_bucket: ggml-org/cache214      #     save: true215 216  gpu-webgpu-nvidia:217    runs-on: "hf-jobs-t4-small:ubuntu26_04"218 219    steps:220      - name: Clone221        id: checkout222        uses: actions/checkout@v6223 224      - name: Install dependencies225        run: |226          sudo apt update227          sudo apt install -y build-essential cmake libxcb-xinput0 libxcb-xinerama0 libxcb-cursor-dev libvulkan1 mesa-vulkan-drivers libglvnd0 libgl1 libglx0 libegl1 libgles2 libssl-dev time unzip wget python3 python3-venv python3-pip228 229      - name: ccache230        uses: ggml-org/ccache-action@v1.2.24231        with:232          restore: false233          save: false234 235      - name: ccache-buckets-restore236        uses: ./.github/actions/ccache-buckets237        with:238          key: self-hosted-webgpu-nvidia239          folder: llama.cpp240          hf_bucket: ggml-org/cache241 242      - name: Dawn Dependency243        id: dawn-depends244        run: |245          DAWN_VERSION="v20260908.214631"246          DAWN_OWNER="google"247          DAWN_REPO="dawn"248          DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-ubuntu-latest-Release"249          echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"250          curl -L -o artifact.tar.gz \251            "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"252          mkdir dawn253          tar -xvf artifact.tar.gz -C dawn --strip-components=1254 255      - name: Test256        id: ggml-ci257        run: |258          GG_BUILD_WEBGPU=1 \259          GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \260          GG_BUILD_WEBGPU_DAWN_DIR="$GITHUB_WORKSPACE/dawn/lib64/cmake/Dawn" \261            bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp262 263      - name: ccache-buckets-save264        if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}265        uses: ./.github/actions/ccache-buckets266        env:267          HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}268        with:269          key: self-hosted-webgpu-nvidia270          folder: llama.cpp271          evict-old-files: 1d272          hf_bucket: ggml-org/cache273          save: true274 275  # TODO: provision AMX-compatible machine276  #cpu-amx:277  #  runs-on: [self-hosted, Linux, CPU, AMX]278 279  #  steps:280  #    - name: Clone281  #      id: checkout282  #      uses: actions/checkout@v6283 284  #    - name: Test285  #      id: ggml-ci286  #      run: |287  #        bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp288 289  # TODO: provision AMD GPU machine290  # amd-vulkan:291  #   runs-on: [self-hosted, Linux, AMD]292 293  #   steps:294  #     - name: Clone295  #       id: checkout296  #       uses: actions/checkout@v6297 298  #     - name: Test299  #       id: ggml-ci300  #       run: |301  #         vulkaninfo --summary302  #         GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp303 304  # TODO: provision AMD GPU machine305  # amd-rocm:306  #   runs-on: [self-hosted, Linux, AMD]307 308  #   steps:309  #     - name: Clone310  #       id: checkout311  #       uses: actions/checkout@v6312 313  #     - name: Test314  #       id: ggml-ci315  #       run: |316  #         amd-smi static317  #         GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS="gfx1101" bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp318 319  gpu-metal:320    runs-on: [self-hosted, macOS, ARM64]321 322    steps:323      - name: Clone324        id: checkout325        uses: actions/checkout@v6326 327      - name: Test328        id: ggml-ci329        run: |330          GG_BUILD_METAL=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp331 332  gpu-webgpu-apple:333    runs-on: [self-hosted, macOS, ARM64]334 335    steps:336      - name: Clone337        id: checkout338        uses: actions/checkout@v6339 340      - name: Dawn Dependency341        id: dawn-depends342        run: |343          DAWN_VERSION="v20260908.214631"344          DAWN_OWNER="google"345          DAWN_REPO="dawn"346          DAWN_ASSET_NAME="Dawn-94c3c9cc0d5fb2e85aebb370fa8d37b71aa34655-macos-latest-Release"347          echo "Fetching release asset from https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"348          curl -L -o artifact.tar.gz \349            "https://github.com/google/dawn/releases/download/${DAWN_VERSION}/${DAWN_ASSET_NAME}.tar.gz"350          mkdir dawn351          tar -xvf artifact.tar.gz -C dawn --strip-components=1352 353      - name: Test354        id: ggml-ci355        run: |356          GG_BUILD_WEBGPU=1 GG_BUILD_WEBGPU_DAWN_PREFIX="$GITHUB_WORKSPACE/dawn" \357            bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp358 359  gpu-vulkan-apple:360    runs-on: [self-hosted, macOS, ARM64]361 362    steps:363      - name: Clone364        id: checkout365        uses: actions/checkout@v6366 367      - name: Test368        id: ggml-ci369        run: |370          vulkaninfo --summary371          GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp372 373  gpu-vulkan-intel-linux:374    runs-on: [self-hosted, Linux, Intel]375 376    steps:377      - name: Clone378        id: checkout379        uses: actions/checkout@v6380        with:381          persist-credentials: false382 383      - name: Test384        id: ggml-ci385        run: |386          vulkaninfo --summary387          GG_BUILD_VULKAN=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp388 389  gpu-vulkan-intel-windows:390    runs-on: [self-hosted, Windows, X64, Intel]391 392    steps:393      - name: Clone394        id: checkout395        uses: actions/checkout@v6396 397      - name: Test398        id: ggml-ci399        shell: C:\msys64\usr\bin\bash.exe --noprofile --norc -eo pipefail "{0}"400        env:401          MSYSTEM: UCRT64402          CHERE_INVOKING: 1403          PATH: C:\msys64\ucrt64\bin;C:\msys64\usr\bin;C:\Windows\System32;${{ env.PATH }}404        run: |405          vulkaninfo --summary406          # Skip python related tests with GG_BUILD_LOW_PERF=1 since Windows MSYS2 UCRT64 currently fails to create407          # a valid python environment for testing408          LLAMA_FATAL_WARNINGS=OFF GG_BUILD_NINJA=1 GG_BUILD_VULKAN=1 GG_BUILD_LOW_PERF=1 ./ci/run.sh ./results/llama.cpp ./mnt/llama.cpp409 410  gpu-openvino-low-perf:411    runs-on: [self-hosted, Linux, Intel, OpenVINO]412 413    env:414      # Sync versions in build.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile415      OPENVINO_VERSION_MAJOR: "2026.4"416      OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"417 418    steps:419      - name: Clone420        id: checkout421        uses: actions/checkout@v6422 423      - name: Setup OpenVINO Toolkit424        uses: ./.github/actions/linux-setup-openvino425        with:426          path: ./openvino_toolkit427          version_major: ${{ env.OPENVINO_VERSION_MAJOR }}428          version_full: ${{ env.OPENVINO_VERSION_FULL }}429 430      - name: Install OpenVINO dependencies431        run: |432          cd ./openvino_toolkit433          chmod +x ./install_dependencies/install_openvino_dependencies.sh434          echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh435 436      - name: Test437        id: ggml-ci438        run: |439          source ./openvino_toolkit/setupvars.sh440          GG_BUILD_OPENVINO=1 GGML_OPENVINO_DEVICE=GPU GG_BUILD_LOW_PERF=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp441 442  cpu-x64-high-perf:443    runs-on: [self-hosted, Linux, X64]444 445    steps:446      - name: Clone447        id: checkout448        uses: actions/checkout@v6449 450      - name: Test451        id: ggml-ci452        run: |453          LLAMA_ARG_THREADS=$(nproc) GG_BUILD_HIGH_PERF=1 GG_BUILD_EXTRA_TESTS_0=1 bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp454 455  cpu-arm64-high-perf-graviton4:456    runs-on: ah-ubuntu_24_04-c8g_8x457 458    steps:459      - name: Clone460        id: checkout461        uses: actions/checkout@v6462 463      - name: Dependencies464        id: depends465        run: |466          set -euxo pipefail467          sudo apt-get update468          sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \469          apt-get install -y \470          build-essential \471          python3-venv \472          gpg \473          wget \474          time \475          git-lfs476 477          git lfs install478 479          # install the latest cmake480          sudo install -d /usr/share/keyrings481          wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \482            | gpg --dearmor \483            | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null484          echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \485            | sudo tee /etc/apt/sources.list.d/kitware.list486          sudo apt-get update487          sudo apt-get install -y cmake488 489      - name: Test490        id: ggml-ci491        run: |492          LLAMA_ARG_THREADS=$(nproc) \493          GG_BUILD_HIGH_PERF=1 \494          GG_BUILD_NO_BF16=1 \495          GG_BUILD_EXTRA_TESTS_0=1 \496          bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp497 498  cpu-arm64-graviton4-kleidiai:499    runs-on: ah-ubuntu_24_04-c8g_8x500 501    steps:502      - name: Clone503        id: checkout504        uses: actions/checkout@v6505 506      - name: Dependencies507        id: depends508        run: |509          set -euxo pipefail510          sudo apt-get update511          sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \512          apt-get install -y \513          build-essential \514          python3-venv \515          gpg \516          wget \517          time \518          git-lfs519 520          git lfs install521 522          # install the latest cmake523          sudo install -d /usr/share/keyrings524          wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \525            | gpg --dearmor \526            | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null527          echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \528            | sudo tee /etc/apt/sources.list.d/kitware.list529          sudo apt-get update530          sudo apt-get install -y cmake531 532      - name: Test533        id: ggml-ci534        run: |535          LLAMA_ARG_THREADS=$(nproc) \536          GG_BUILD_KLEIDIAI=1 \537          GG_BUILD_EXTRA_TESTS_0=1 \538          GG_BUILD_HIGH_PERF=1 \539          bash ./ci/run.sh ~/results/llama.cpp ~/mnt/llama.cpp540