Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
release.yml1922 linesDownload Raw Back to workflows
1name: Release2 3on:4  workflow_dispatch: # allows manual triggering5    inputs:6      create_release:7        description: 'Create new release'8        required: true9        type: boolean10  push:11    branches:12      - master13    paths: [14      '.github/workflows/release.yml',15      '**/CMakeLists.txt',16      '**/.cmake',17      '**/*.h',18      '**/*.hpp',19      '**/*.c',20      '**/*.cpp',21      '**/*.cu',22      '**/*.cuh',23      '**/*.swift',24      '**/*.m',25      '**/*.metal',26      '**/*.comp',27      '**/*.glsl'28    ]29 30env:31  GH_TOKEN: ${{ github.token }}32  BRANCH_NAME: ${{ github.head_ref || github.ref_name }}33  CMAKE_ARGS: "-DLLAMA_BUILD_EXAMPLES=OFF -DLLAMA_BUILD_TESTS=OFF -DLLAMA_BUILD_TOOLS=ON -DLLAMA_BUILD_SERVER=ON -DGGML_RPC=ON"34 35# note: run this workflow one at a time for better cache reuse36concurrency:37  group: release38  queue: max39 40jobs:41  check-release:42    runs-on: ubuntu-slim43 44    outputs:45      should_release: ${{ steps.check.outputs.should_release }}46 47    steps:48      - id: check49        env:50          COMMIT_MESSAGE: ${{ github.event.head_commit.message }}51        run: |52          if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then53            echo "should_release=true" >> $GITHUB_OUTPUT54          elif [[ "${{ github.event_name }}" == "push" && "${{ github.ref }}" == "refs/heads/master" ]]; then55            if echo "$COMMIT_MESSAGE" | grep -q '\[no release\]'; then56              echo "should_release=false" >> $GITHUB_OUTPUT57            else58              echo "should_release=true" >> $GITHUB_OUTPUT59            fi60          else61            echo "should_release=false" >> $GITHUB_OUTPUT62          fi63 64  macos-cpu:65    needs: [check-release, ui-build]66    if: ${{ needs.check-release.outputs.should_release == 'true' }}67    strategy:68      matrix:69        include:70          - build: 'arm64'71            arch: 'arm64'72            os: macos-2673            defines: "-DGGML_METAL_EMBED_LIBRARY=ON -DCMAKE_OSX_DEPLOYMENT_TARGET=13.3"74          # TODO: this build is disabled to save Github Actions resources (https://github.com/ggml-org/llama.cpp/pull/23780)75          #       in order to enable it again, we have to provision dedicated runners  to run it76          #- build: 'arm64-kleidiai'77          #  arch: 'arm64'78          #  os: macos-1479          #  defines: "-DGGML_METAL_EMBED_LIBRARY=ON -DCMAKE_OSX_DEPLOYMENT_TARGET=13.3 -DGGML_CPU_KLEIDIAI=ON"80          - build: 'x64'81            arch: 'x64'82            os: macos-15-intel83            # Metal is disabled on x64 due to intermittent failures with Github runners not having a GPU:84            # https://github.com/ggml-org/llama.cpp/actions/runs/8635935781/job/23674807267#step:5:231385            defines: "-DGGML_METAL=OFF -DCMAKE_OSX_DEPLOYMENT_TARGET=13.3"86 87    runs-on: ${{ matrix.os }}88 89    permissions:90      actions: write91 92    steps:93      - name: Clone94        id: checkout95        uses: actions/checkout@v696        with:97          fetch-depth: 098 99      - name: Download UI build100        uses: actions/download-artifact@v7101        with:102          name: llama-ui.zip103          path: tools/ui/dist104 105      - name: ccache106        uses: ggml-org/ccache-action@v1.2.24107        with:108          key: release-${{ matrix.os }}-${{ matrix.arch }}109          evict-old-files: 1d110 111      - name: Build112        id: cmake_build113        run: |114          sysctl -a115          cmake -B build \116            ${{ matrix.defines }} \117            -DCMAKE_INSTALL_RPATH='@loader_path' \118            -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \119            -DLLAMA_FATAL_WARNINGS=ON \120            -DLLAMA_BUILD_BORINGSSL=ON \121            ${{ env.CMAKE_ARGS }}122          cmake --build build --config Release -j $(sysctl -n hw.logicalcpu)123 124      - name: Determine tag name125        id: tag126        uses: ./.github/actions/get-tag-name127 128      - name: Pack artifacts129        id: pack_artifacts130        run: |131          cp LICENSE ./build/bin/132          tar -czvf llama-${{ steps.tag.outputs.name }}-bin-macos-${{ matrix.build }}.tar.gz -s ",^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .133 134      - name: Upload artifacts135        uses: actions/upload-artifact@v6136        with:137          path: llama-${{ steps.tag.outputs.name }}-bin-macos-${{ matrix.build }}.tar.gz138          name: llama-bin-macos-${{ matrix.build }}.tar.gz139 140      - name: ccache-clear141        uses: ./.github/actions/ccache-clear142        with:143          key: release-${{ matrix.os }}-${{ matrix.arch }}144 145  ubuntu-cpu:146    needs: [check-release, ui-build]147    if: ${{ needs.check-release.outputs.should_release == 'true' }}148    strategy:149      matrix:150        include:151          - build: 'x64'152            os: ubuntu-22.04153          - build: 'arm64'154            os: ubuntu-24.04-arm155          - build: 's390x'156            os: ubuntu-24.04-s390x157 158    runs-on: ${{ matrix.os }}159 160    permissions:161      actions: write162 163    steps:164      - name: Clone165        id: checkout166        uses: actions/checkout@v6167        with:168          fetch-depth: 0169 170      - name: Download UI build171        uses: actions/download-artifact@v7172        with:173          name: llama-ui.zip174          path: tools/ui/dist175 176      - name: Dependencies177        id: depends178        run: |179          sudo apt-get update180          sudo apt-get install build-essential libssl-dev181 182      - name: Toolchain workaround (GCC 14)183        if: ${{ contains(matrix.os, 'ubuntu-24.04') }}184        run: |185          sudo apt-get install -y gcc-14 g++-14186          echo "CC=gcc-14" >> "$GITHUB_ENV"187          echo "CXX=g++-14" >> "$GITHUB_ENV"188 189      - name: ccache190        if: ${{ matrix.build != 's390x' }}191        uses: ggml-org/ccache-action@v1.2.24192        with:193          key: release-${{ matrix.os }}-cpu194          evict-old-files: 1d195 196      - name: Build197        id: cmake_build198        run: |199          cmake -B build \200            -DCMAKE_INSTALL_RPATH='$ORIGIN' \201            -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \202            -DGGML_BACKEND_DL=ON \203            -DGGML_NATIVE=OFF \204            -DGGML_CPU_ALL_VARIANTS=ON \205            -DLLAMA_FATAL_WARNINGS=ON \206            ${{ env.CMAKE_ARGS }}207          cmake --build build --config Release -j $(nproc)208 209      - name: Determine tag name210        id: tag211        uses: ./.github/actions/get-tag-name212 213      - name: Pack artifacts214        id: pack_artifacts215        run: |216          cp LICENSE ./build/bin/217          tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .218 219      - name: Upload artifacts220        uses: actions/upload-artifact@v6221        with:222          path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-${{ matrix.build }}.tar.gz223          name: llama-bin-ubuntu-${{ matrix.build }}.tar.gz224 225      - name: ccache-clear226        if: ${{ matrix.build != 's390x' }}227        uses: ./.github/actions/ccache-clear228        with:229          key: release-${{ matrix.os }}-cpu230 231  ubuntu-vulkan:232    needs: [check-release, ui-build]233    if: ${{ needs.check-release.outputs.should_release == 'true' }}234 235    strategy:236      matrix:237        include:238          - build: 'x64'239            os: ubuntu-22.04240          - build: 'arm64'241            os: ubuntu-24.04-arm242 243    runs-on: ${{ matrix.os }}244 245    permissions:246      actions: write247 248    steps:249      - name: Clone250        id: checkout251        uses: actions/checkout@v6252        with:253          fetch-depth: 0254 255      - name: Download UI build256        uses: actions/download-artifact@v7257        with:258          name: llama-ui.zip259          path: tools/ui/dist260 261      - name: Dependencies262        id: depends263        run: |264          if [[ "${{ matrix.os }}" =~ "ubuntu-22.04" ]]; then265            wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | sudo apt-key add -266            sudo wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list267            sudo apt-get update -y268            sudo apt-get install -y build-essential mesa-vulkan-drivers vulkan-sdk libssl-dev269          else270            sudo apt-get update -y271            sudo apt-get install -y gcc-14 g++-14 build-essential glslc libvulkan-dev spirv-headers libssl-dev ninja-build272            echo "CC=gcc-14" >> "$GITHUB_ENV"273            echo "CXX=g++-14" >> "$GITHUB_ENV"274          fi275 276      - name: ccache277        uses: ggml-org/ccache-action@v1.2.24278        with:279          key: release-${{ matrix.os }}-vulkan280          evict-old-files: 1d281 282      - name: Build283        id: cmake_build284        run: |285          cmake -B build \286            -DCMAKE_INSTALL_RPATH='$ORIGIN' \287            -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \288            -DGGML_BACKEND_DL=ON \289            -DGGML_NATIVE=OFF \290            -DGGML_CPU_ALL_VARIANTS=ON \291            -DGGML_VULKAN=ON \292            ${{ env.CMAKE_ARGS }}293          cmake --build build --config Release -j $(nproc)294 295      - name: Determine tag name296        id: tag297        uses: ./.github/actions/get-tag-name298 299      - name: Pack artifacts300        id: pack_artifacts301        run: |302          cp LICENSE ./build/bin/303          tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .304 305      - name: Upload artifacts306        uses: actions/upload-artifact@v6307        with:308          path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz309          name: llama-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz310 311      - name: ccache-clear312        uses: ./.github/actions/ccache-clear313        with:314          key: release-${{ matrix.os }}-vulkan315 316  ubuntu-cuda:317    name: ubuntu-cuda (${{ matrix.label }}, ${{ matrix.build }})318    needs: [check-release, ui-build]319    if: ${{ needs.check-release.outputs.should_release == 'true' }}320 321    strategy:322      matrix:323        include:324          # label = short version used in artifact names / release body325          # cuda  = full container image tag326          - build: 'x64'327            os: ubuntu-24.04328            cuda: '12.8.2'329            label: '12.8'330            defines: '-DGGML_CUDA_CUB_3DOT2=ON'331          - build: 'x64'332            os: ubuntu-24.04333            cuda: '13.3.1'334            label: '13.3'335            defines: ''336          - build: 'arm64'337            os: ubuntu-24.04-arm338            cuda: '13.3.1'339            label: '13.3'340            defines: ''341 342    runs-on: ${{ matrix.os }}343    container: nvidia/cuda:${{ matrix.cuda }}-devel-ubuntu24.04344 345    permissions:346      actions: write347 348    steps:349      # the container has no git; install it before checkout so that a real git350      # repository is created (the get-tag-name action and the build both need it)351      - name: Install git352        run: |353          apt-get update354          apt-get install -y --no-install-recommends git355 356      - name: Clone357        id: checkout358        uses: actions/checkout@v6359        with:360          fetch-depth: 0361 362      # checkout runs as the host user; in-container steps run as root, so git363      # refuses to touch a repo it does not own. Mark the workspace as safe.364      # use the env var: the github.workspace context holds the HOST path,365      # GITHUB_WORKSPACE the container path366      - name: Git safe directory367        run: git config --global --add safe.directory "$GITHUB_WORKSPACE"368 369      - name: Download UI build370        uses: actions/download-artifact@v7371        with:372          name: llama-ui.zip373          path: tools/ui/dist374 375      - name: Dependencies376        id: depends377        # container jobs default to sh (dash); need bash for the [[ ]] below378        shell: bash379        run: |380          apt-get update381          apt-get install -y --no-install-recommends build-essential cmake ninja-build libssl-dev jq python3-venv382          # the container ships GCC 13, which does not know the 'sme' march383          # feature used by the armv9.2 CPU variant of GGML_CPU_ALL_VARIANTS384          if [[ "${{ matrix.build }}" == "arm64" ]]; then385            apt-get install -y --no-install-recommends gcc-14 g++-14386            echo "CC=gcc-14" >> "$GITHUB_ENV"387            echo "CXX=g++-14" >> "$GITHUB_ENV"388          fi389 390      - name: ccache391        uses: ggml-org/ccache-action@v1.2.24392        with:393          key: release-ubuntu-${{ matrix.os }}-cuda-${{ matrix.label }}-${{ matrix.build }}394          evict-old-files: 1d395          max-size: "1G"396 397      - name: Build398        id: cmake_build399        # no CMAKE_CUDA_ARCHITECTURES: use the broad default arch set from400        # ggml/src/ggml-cuda/CMakeLists.txt so the release binary covers many GPUs401        run: |402          cmake -B build \403            -DCMAKE_INSTALL_RPATH='$ORIGIN' \404            -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \405            -DGGML_BACKEND_DL=ON \406            -DGGML_NATIVE=OFF \407            -DGGML_CPU_ALL_VARIANTS=ON \408            -DGGML_CUDA=ON \409            -DGGML_CUDA_NCCL=OFF \410            ${{ env.CMAKE_ARGS }} ${{ matrix.defines }}411          cmake --build build --config Release -j $(nproc)412 413      - name: Determine tag name414        id: tag415        uses: ./.github/actions/get-tag-name416 417      - name: Pack artifacts418        id: pack_artifacts419        run: |420          cp LICENSE ./build/bin/421          tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .422 423      - name: Upload artifacts424        uses: actions/upload-artifact@v6425        with:426          path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz427          name: llama-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz428 429      # ship the CUDA runtime libraries the backend links against, mirroring430      # the windows-cuda cudart zip - extract next to the binaries ($ORIGIN rpath)431      - name: Pack CUDA runtime432        id: pack_cuda_runtime433        run: |434          major="${{ matrix.label }}"435          major="${major%%.*}"436          mkdir -p ./cudart437          # cp -L dereferences the SONAME symlinks into plain files, so the438          # tarball holds exactly 3 files with no versioned duplicates439          cp -L /usr/local/cuda/lib64/libcudart.so.${major} ./cudart/440          cp -L /usr/local/cuda/lib64/libcublas.so.${major} ./cudart/441          cp -L /usr/local/cuda/lib64/libcublasLt.so.${major} ./cudart/442          tar -czvf cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}," -C ./cudart .443 444      - name: Upload CUDA runtime445        uses: actions/upload-artifact@v6446        with:447          path: cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz448          name: cudart-llama-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz449 450      - name: ccache-clear451        uses: ./.github/actions/ccache-clear452        with:453          key: release-ubuntu-${{ matrix.os }}-cuda-${{ matrix.label }}-${{ matrix.build }}454 455  android-arm64:456    needs: [check-release, ui-build]457    if: ${{ needs.check-release.outputs.should_release == 'true' }}458 459    runs-on: ubuntu-24.04  # previously ubuntu-latest460 461    #permissions:462    #  actions: write463 464    env:465      NDK_VERSION: "29.0.14206865"466 467    steps:468      - name: Clone469        id: checkout470        uses: actions/checkout@v6471        with:472          fetch-depth: 0473 474      - name: Download UI build475        uses: actions/download-artifact@v7476        with:477          name: llama-ui.zip478          path: tools/ui/dist479 480      - name: Set up JDK481        uses: actions/setup-java@v5482        with:483          java-version: 17484          distribution: temurin485 486      - name: Setup Android SDK487        uses: android-actions/setup-android@be39fa834029ff78f1a44aa3bb0819b8fc2bd8fd # v4.0.4488        with:489          log-accepted-android-sdk-licenses: false490 491      - name: Install NDK492        run: |493          sdkmanager "ndk;${{ env.NDK_VERSION }}"494          echo "ANDROID_NDK=${ANDROID_SDK_ROOT}/ndk/${{ env.NDK_VERSION }}" >> $GITHUB_ENV495 496      # note : disabled to spare some cache space (https://github.com/ggml-org/llama.cpp/pull/23789)497      #        for some reason, the ccache does not improve the build time in this case498      # example:499      #   cache off: https://github.com/ggerganov/tmp2/actions/runs/26534713799/job/78160400831500      #   cache on:  https://github.com/ggerganov/tmp2/actions/runs/26534713799/job/78224189394501      #502      #- name: ccache503      #  uses: ggml-org/ccache-action@v1.2.24504      #  with:505      #    key: release-android-arm64506      #    evict-old-files: 1d507 508      - name: Build509        id: cmake_build510        run: |511          cmake -B build \512            -DCMAKE_TOOLCHAIN_FILE=${ANDROID_NDK}/build/cmake/android.toolchain.cmake \513            -DANDROID_ABI=arm64-v8a \514            -DANDROID_PLATFORM=android-28 \515            -DCMAKE_INSTALL_RPATH='$ORIGIN' \516            -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \517            -DGGML_BACKEND_DL=ON \518            -DGGML_NATIVE=OFF \519            -DGGML_CPU_ALL_VARIANTS=ON \520            -DLLAMA_FATAL_WARNINGS=ON \521            -DGGML_OPENMP=OFF \522            -DLLAMA_BUILD_BORINGSSL=ON \523            ${{ env.CMAKE_ARGS }}524          cmake --build build --config Release -j $(nproc)525 526      #- name: ccache-clear527      #  uses: ./.github/actions/ccache-clear528      #  with:529      #    key: release-android-arm64530 531      - name: Determine tag name532        id: tag533        uses: ./.github/actions/get-tag-name534 535      - name: Pack artifacts536        id: pack_artifacts537        run: |538          cp LICENSE ./build/bin/539          tar -czvf llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .540 541      - name: Upload artifacts542        uses: actions/upload-artifact@v6543        with:544          path: llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz545          name: llama-bin-android-arm64.tar.gz546 547  ubuntu-24-openvino:548    needs: [check-release, ui-build]549    if: ${{ needs.check-release.outputs.should_release == 'true' }}550 551    runs-on: ubuntu-24.04552 553    permissions:554      actions: write555 556    outputs:557      openvino_version: ${{ steps.openvino_version.outputs.value }}558 559    env:560      # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile561      OPENVINO_VERSION_MAJOR: "2026.4"562      OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"563 564    steps:565      - name: Set OpenVINO version output566        id: openvino_version567        run: echo "value=${{ env.OPENVINO_VERSION_MAJOR }}" >> $GITHUB_OUTPUT568 569      - name: Clone570        id: checkout571        uses: actions/checkout@v6572        with:573          fetch-depth: 0574 575      - name: Download UI build576        uses: actions/download-artifact@v7577        with:578          name: llama-ui.zip579          path: tools/ui/dist580 581      - name: ccache582        uses: ggml-org/ccache-action@v1.2.24583        with:584          key: release-ubuntu-24.04-openvino-release-no-preset-v1585          evict-old-files: 1d586 587      - name: Dependencies588        run: |589          sudo apt-get update590          sudo apt-get install -y build-essential libssl-dev libtbb12 cmake ninja-build python3-pip591          sudo apt install ocl-icd-opencl-dev opencl-headers opencl-clhpp-headers intel-opencl-icd592 593      - name: Use OpenVINO Toolkit Cache594        uses: actions/cache@v5595        id: cache-openvino596        with:597          path: ./openvino_toolkit598          key: cache-gha-openvino-toolkit-v${{ env.OPENVINO_VERSION_FULL }}-${{ runner.os }}599 600      - name: Setup OpenVINO Toolkit601        if: steps.cache-openvino.outputs.cache-hit != 'true'602        uses: ./.github/actions/linux-setup-openvino603        with:604          path: ./openvino_toolkit605          version_major: ${{ env.OPENVINO_VERSION_MAJOR }}606          version_full: ${{ env.OPENVINO_VERSION_FULL }}607 608      - name: Install OpenVINO dependencies609        run: |610          cd ./openvino_toolkit611          chmod +x ./install_dependencies/install_openvino_dependencies.sh612          echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh613 614      - name: Build615        id: cmake_build616        run: |617          source ./openvino_toolkit/setupvars.sh618          cmake -B build/ReleaseOV -G Ninja \619            -DCMAKE_BUILD_TYPE=Release \620            -DGGML_OPENVINO=ON \621            -DCMAKE_INSTALL_RPATH='$ORIGIN' \622            -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \623            ${{ env.CMAKE_ARGS }}624          cmake --build build/ReleaseOV --config Release --parallel625 626      - name: Determine tag name627        id: tag628        uses: ./.github/actions/get-tag-name629 630      - name: Pack artifacts631        id: pack_artifacts632        run: |633          dest=./build/ReleaseOV/bin634          OPENVINO_ROOT=./openvino_toolkit635          ov_lib="$OPENVINO_ROOT/runtime/lib/intel64"636 637          # Bundle OpenVINO runtime libs + TBB. Binaries built with RPATH=$ORIGIN638          # load these siblings without setupvars.sh / LD_LIBRARY_PATH.639          cp -P "$ov_lib"/libopenvino.so* \640                "$ov_lib"/libopenvino_c.so* \641                "$ov_lib"/libopenvino_*_plugin.so \642                "$ov_lib"/libopenvino_intel_npu_compiler*.so \643                "$OPENVINO_ROOT"/runtime/3rdparty/tbb/lib/*.so* \644                "$dest"645          cp -P /usr/lib/x86_64-linux-gnu/libOpenCL.so.1* "$dest" 2>/dev/null || true646          cp "$ov_lib"/cache.json "$dest" 2>/dev/null || true647 648          # OpenVINO licensing649          cp -r "$OPENVINO_ROOT"/docs/licensing "$dest"/openvino-licensing650 651          cp LICENSE "$dest"652          tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C "$dest" .653 654      - name: Upload artifacts655        uses: actions/upload-artifact@v6656        with:657          path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz658          name: llama-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz659 660      - name: ccache-clear661        uses: ./.github/actions/ccache-clear662        with:663          key: release-ubuntu-24.04-openvino-release-no-preset-v1664 665  windows-openvino:666    needs: [check-release, ui-build]667    if: ${{ needs.check-release.outputs.should_release == 'true' }}668 669    runs-on: windows-2022670 671    outputs:672      openvino_version: ${{ steps.openvino_version.outputs.value }}673 674    env:675      # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile676      OPENVINO_VERSION_MAJOR: "2026.4"677      OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"678 679    steps:680      - name: Set OpenVINO version output681        id: openvino_version682        shell: bash683        run: echo "value=${{ env.OPENVINO_VERSION_MAJOR }}" >> $GITHUB_OUTPUT684 685      - name: Clone686        id: checkout687        uses: actions/checkout@v6688        with:689            fetch-depth: 0690 691      - name: Download UI build692        uses: actions/download-artifact@v7693        with:694          name: llama-ui.zip695          path: tools/ui/dist696 697      - name: ccache698        uses: ggml-org/ccache-action@v1.2.24699        with:700          key: release-windows-2022-openvino701          variant: ccache702          evict-old-files: 1d703 704      - name: Setup Cache705        uses: actions/cache@v5706        id: cache-openvino707        with:708          path: ./openvino_toolkit709          key: cache-gha-openvino-toolkit-v${{ env.OPENVINO_VERSION_FULL }}-${{ runner.os }}710 711      - name: Setup OpenVINO Toolkit712        if: steps.cache-openvino.outputs.cache-hit != 'true'713        uses: ./.github/actions/windows-setup-openvino714        with:715          path: ./openvino_toolkit716          version_major: ${{ env.OPENVINO_VERSION_MAJOR }}717          version_full: ${{ env.OPENVINO_VERSION_FULL }}718 719      - name: Install OpenCL using vcpkg720        shell: powershell721        run: |722          git clone https://github.com/microsoft/vcpkg C:\vcpkg723          C:\vcpkg\bootstrap-vcpkg.bat724          C:\vcpkg\vcpkg install opencl725 726      - name: Build727        id: cmake_build728        shell: cmd729        run: |730          REM Find extracted OpenVINO folder dynamically731          for /d %%i in (openvino_toolkit\*) do set OPENVINO_ROOT=%%i732 733          if not exist "%OPENVINO_ROOT%\runtime\cmake\OpenVINOConfig.cmake" (734              echo ERROR: OpenVINOConfig.cmake not found735              exit /b 1736          )737 738          call "%OPENVINO_ROOT%\setupvars.bat"739 740          cmake -B build\ReleaseOV -G "Visual Studio 17 2022" ^741            -A x64 ^742            -DCMAKE_BUILD_TYPE=Release ^743            -DGGML_OPENVINO=ON ^744            -DLLAMA_BUILD_BORINGSSL=ON ^745            -DCMAKE_TOOLCHAIN_FILE=C:\vcpkg\scripts\buildsystems\vcpkg.cmake ^746            ${{ env.CMAKE_ARGS }}747 748          cmake --build build\ReleaseOV --config Release -- /m749 750      - name: Determine tag name751        id: tag752        uses: ./.github/actions/get-tag-name753 754      - name: Pack artifacts755        id: pack_artifacts756        shell: powershell757        run: |758          # Locate the extracted OpenVINO toolkit root (same pattern as the Build step).759          $OPENVINO_ROOT = (Get-ChildItem -Directory openvino_toolkit | Select-Object -First 1).FullName760          if (-not $OPENVINO_ROOT) {761            Write-Error "OpenVINO toolkit folder not found under .\openvino_toolkit"762            exit 1763          }764 765          $dest = ".\build\ReleaseOV\bin\Release"766 767          $ovBin = Join-Path $OPENVINO_ROOT 'runtime\bin\intel64\Release'768          Copy-Item -Path (Join-Path $ovBin '*.dll')       -Destination $dest -Force769          Copy-Item -Path (Join-Path $ovBin 'cache.json')  -Destination $dest -Force770 771          $tbbBin = Join-Path $OPENVINO_ROOT 'runtime\3rdparty\tbb\bin'772          Copy-Item -Path (Join-Path $tbbBin 'tbb*.dll') -Destination $dest -Force773 774          # OpenVINO licensing775          $licensingDest = Join-Path $dest 'openvino-licensing'776          New-Item -ItemType Directory -Force -Path $licensingDest | Out-Null777          Copy-Item -Path (Join-Path $OPENVINO_ROOT 'docs\licensing\*') -Destination $licensingDest -Recurse -Force778 779          Copy-Item LICENSE $dest780          7z a -snl llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip $dest\*781 782      - name: Upload artifacts783        uses: actions/upload-artifact@v6784        with:785          path: llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip786          name: llama-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip787 788      - name: ccache-clear789        uses: ./.github/actions/ccache-clear790        with:791          key: release-windows-2022-openvino792 793  windows-cpu:794    name: windows-cpu / ${{ matrix.arch }}795    needs: [check-release, ui-build]796    if: ${{ needs.check-release.outputs.should_release == 'true' }}797 798    runs-on: windows-2025-vs2026799 800    permissions:801      actions: write802 803    strategy:804      matrix:805        include:806          - arch: 'x64'807          - arch: 'arm64'808 809    steps:810      - name: Clone811        uses: actions/checkout@v6812        with:813          fetch-depth: 0814 815      - name: Download UI build816        uses: actions/download-artifact@v7817        with:818          name: llama-ui.zip819          path: tools/ui/dist820 821      - name: Install Ninja822        run: |823          choco install ninja824 825      - name: ccache826        uses: ggml-org/ccache-action@v1.2.24827        with:828          key: release-windows-2025-vs2026-${{ matrix.arch }}-cpu829          evict-old-files: 1d830 831      - name: Build832        shell: cmd833        run: |834          call "C:\Program Files\Microsoft Visual Studio\18\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }}835          cmake -S . -B build -G "Ninja Multi-Config" ^836            -D CMAKE_TOOLCHAIN_FILE=cmake/${{ matrix.arch }}-windows-llvm.cmake ^837            -DLLAMA_BUILD_BORINGSSL=ON ^838            -DGGML_NATIVE=OFF ^839            -DGGML_BACKEND_DL=ON ^840            -DGGML_CPU_ALL_VARIANTS=${{ matrix.arch == 'x64' && 'ON' || 'OFF' }} ^841            -DGGML_OPENMP=ON ^842            -DGGML_OPENMP_FETCH=ON ^843            ${{ env.CMAKE_ARGS }}844          cmake --build build --config Release845 846      - name: Pack artifacts847        id: pack_artifacts848        run: |849          7z a -snl llama-bin-win-cpu-${{ matrix.arch }}.zip .\build\bin\Release\*850 851      - name: Upload artifacts852        uses: actions/upload-artifact@v6853        with:854          path: llama-bin-win-cpu-${{ matrix.arch }}.zip855          name: llama-bin-win-cpu-${{ matrix.arch }}.zip856 857      - name: ccache-clear858        uses: ./.github/actions/ccache-clear859        with:860          key: release-windows-2025-vs2026-${{ matrix.arch }}-cpu861 862  # note: builds only the ggml-hip backend - llama-server is injected from the863  #       windows-cpu zip during the release "Merge artifacts" step864  windows-rocm:865    needs: [check-release]866    if: ${{ needs.check-release.outputs.should_release == 'true' }}867 868    runs-on: windows-2022869 870    strategy:871      matrix:872        include:873          - ROCM_VERSION: "10.0.0"874            gpu_targets: "gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201"875            build: x64876 877    steps:878      - name: Clone879        id: checkout880        uses: actions/checkout@v6881        with:882          fetch-depth: 0883 884      - name: Install Ninja885        run: |886          choco install ninja887 888      - name: ccache889        uses: ggml-org/ccache-action@v1.2.24890        with:891          key: windows-rocm-${{ matrix.ROCM_VERSION }}-${{ matrix.build }}892          evict-old-files: 1d893          max-size: "1G"894 895      # - name: Cache ROCm Installation896      #   id: cache-rocm897      #   uses: actions/cache@v5898      #   with:899      #     path: C:\TheRock\build900      #     key: rocm-wheels-${{ matrix.ROCM_VERSION }}-multi-arch-${{ runner.os }}901 902      - name: Setup ROCm903        # if: steps.cache-rocm.outputs.cache-hit != 'true'904        uses: ./.github/actions/windows-setup-rocm905        with:906          version: ${{ matrix.ROCM_VERSION }}907 908      - name: Setup ROCm Environment909        run: |910          $ErrorActionPreference = "Stop"911 912          # Activate venv from cache or fresh install913          & C:\TheRock\build\.venv\Scripts\Activate.ps1914 915          # Expand the devel tree (idempotent; no-op if already done during install)916          rocm-sdk init917          if ($LASTEXITCODE -ne 0) { throw "rocm-sdk init failed with exit code $LASTEXITCODE" }918 919          # Get ROCm installation paths using the rocm-sdk CLI tool920          $rocmPath = (rocm-sdk path --root)921          if (-not $rocmPath) { throw "rocm-sdk path --root returned empty - devel package may not be installed" }922          $rocmPath = $rocmPath.Trim()923          $cmakePath = (rocm-sdk path --cmake).Trim()924          $binPath = (rocm-sdk path --bin).Trim()925          write-host "ROCm root: $rocmPath"926          write-host "CMake path: $cmakePath"927          write-host "Bin path: $binPath"928 929          echo "HIP_PATH=$rocmPath" >> $env:GITHUB_ENV930          echo "CMAKE_PREFIX_PATH=$cmakePath" >> $env:GITHUB_ENV931          echo "HIP_DEVICE_LIB_PATH=$rocmPath\lib\llvm\amdgcn\bitcode" >> $env:GITHUB_ENV932          echo "HIP_PLATFORM=amd" >> $env:GITHUB_ENV933          echo "LLVM_PATH=$rocmPath\lib\llvm" >> $env:GITHUB_ENV934          echo "$binPath" >> $env:GITHUB_PATH935 936          # Keep venv in PATH for subsequent steps937          echo "C:\TheRock\build\.venv\Scripts" >> $env:GITHUB_PATH938 939      - name: Build940        run: |941          cmake -S . -B build `942            -G "Ninja Multi-Config" `943            -DCMAKE_PREFIX_PATH="${env:HIP_PATH}" `944            -DGGML_BACKEND_DL=ON `945            -DGGML_NATIVE=OFF `946            -DGGML_CPU=OFF `947            -DGGML_HIP=ON `948            -DCMAKE_C_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang.exe" `949            -DCMAKE_CXX_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang++.exe" `950            -DCMAKE_C_FLAGS="-Wno-error=incompatible-pointer-types" `951            -DCMAKE_HIP_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang.exe" `952            -DHIP_PATH="${env:HIP_PATH}" `953            -DAMDGPU_TARGETS="${{ matrix.gpu_targets }}"954          cmake --build build --config Release --parallel ${env:NUMBER_OF_PROCESSORS} --target ggml-hip955 956      - name: Verify HIP backend was built957        run: |958          $hipDll = Get-ChildItem -Path build\bin\Release -Filter "ggml-hip*.dll" -ErrorAction SilentlyContinue959          if (-not $hipDll) {960            Write-Host "##[error]ggml-hip*.dll was NOT produced. The HIP backend silently failed to build."961            Write-Host "Contents of build\bin\Release:"962            Get-ChildItem build\bin\Release | Format-Table -AutoSize963            exit 1964          }965          Write-Host "HIP backend artifact found:"966          $hipDll | Format-Table FullName, Length -AutoSize967 968      - name: Determine tag name969        id: tag970        uses: ./.github/actions/get-tag-name971 972      - name: Get ROCm short version973        run: |974          $rocmVersionShort = ('${{ matrix.ROCM_VERSION }}'.Split('.')[0..1] -join '.')975          echo "ROCM_VERSION_SHORT=$rocmVersionShort" >> $env:GITHUB_ENV976 977      - name: Bundle HIP runtime DLLs (amdhip64_7.dll, rocm_kpack.dll, amd_comgr.dll)978        run: |979          $ErrorActionPreference = "Stop"980          # See issue https://github.com/ggml-org/llama.cpp/issues/26929.981          # ggml-hip.dll loads amdhip64_7.dll at run time. The Adrenalin driver982          # ships an amdhip64_7.dll in System32, which the loader searches before PATH,983          # so a matching DLL from PATH cannot win. Copy amdhip64 next to the984          # binaries (exe directory is searched before System32) so the correct985          # runtime is used. rocm_kpack.dll is amdhip64_7's direct dependency, so986          # copy the matching version too. amd_comgr is copied as well to keep it987          # in sync with the bundled amdhip64, avoiding a version mismatch with a988          # amd_comgr from System32.989          # rocblas/hipblaslt kernels resolve fine via PATH and are not copied.990          $binPath = (rocm-sdk path --bin).Trim()991          if (-not $binPath) { throw "rocm-sdk path --bin returned empty" }992          write-host "ROCm bin path: $binPath"993 994          $patterns = @("amdhip64_7.dll", "rocm_kpack.dll", "amd_comgr.dll")995          foreach ($pattern in $patterns) {996            $files = Get-ChildItem -Path $binPath -Filter $pattern -ErrorAction SilentlyContinue997            if (-not $files) { throw "no match for $pattern in $binPath" }998            foreach ($f in $files) {999              Copy-Item $f.FullName -Destination build\bin\Release -Force1000              write-host "  copied $($f.Name)"1001            }1002          }1003 1004      - name: Pack artifacts1005        run: |1006          7z a -snl llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip `1007            .\build\bin\Release\ggml-hip.dll `1008            .\build\bin\Release\amdhip64_7.dll `1009            .\build\bin\Release\rocm_kpack.dll `1010            .\build\bin\Release\amd_comgr.dll1011 1012      - name: Upload artifacts1013        uses: actions/upload-artifact@v61014        with:1015          path: llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip1016          name: llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip1017 1018      - name: ccache-clear1019        uses: ./.github/actions/ccache-clear1020        with:1021          key: windows-rocm-${{ matrix.ROCM_VERSION }}-${{ matrix.build }}1022 1023  # note: builds only the backend library - llama-server (with the embedded UI)1024  #       is injected from the windows-cpu zip during the release "Merge artifacts" step1025  windows:1026    needs: [check-release]1027    if: ${{ needs.check-release.outputs.should_release == 'true' }}1028 1029    runs-on: windows-20251030 1031    permissions:1032      actions: write1033 1034    env:1035      OPENBLAS_VERSION: 0.3.231036      VULKAN_VERSION: 1.4.357.01037 1038    strategy:1039      matrix:1040        include:1041          - backend: 'vulkan'1042            arch: 'x64'1043            defines: '-DGGML_VULKAN=ON'1044            target: 'ggml-vulkan'1045          - backend: 'opencl-adreno'1046            arch: 'arm64'1047            defines: '-G "Ninja Multi-Config" -D CMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-llvm.cmake -DCMAKE_PREFIX_PATH="$env:RUNNER_TEMP/opencl-arm64-release" -DGGML_OPENCL=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON'1048            target: 'ggml-opencl'1049 1050    steps:1051      - name: Clone1052        id: checkout1053        uses: actions/checkout@v61054 1055      - name: Install Vulkan SDK1056        id: get_vulkan1057        if: ${{ matrix.backend == 'vulkan' }}1058        run: |1059          curl.exe -o $env:RUNNER_TEMP/VulkanSDK-Installer.exe -L "https://sdk.lunarg.com/sdk/download/${env:VULKAN_VERSION}/windows/vulkansdk-windows-X64-${env:VULKAN_VERSION}.exe"1060          & "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install1061          Add-Content $env:GITHUB_ENV "VULKAN_SDK=C:\VulkanSDK\${env:VULKAN_VERSION}"1062          Add-Content $env:GITHUB_PATH "C:\VulkanSDK\${env:VULKAN_VERSION}\bin"1063 1064      - name: Install Ninja1065        id: install_ninja1066        run: |1067          choco install ninja1068 1069      # TODO: these jobs need to use llvm toolchain in order to utilize the ccache1070      #- name: ccache1071      #  uses: ggml-org/ccache-action@v1.2.241072      #  with:1073      #    key: release-windows-2025-${{ matrix.arch }}-${{ matrix.backend }}1074      #    evict-old-files: 1d1075 1076      - name: Install OpenCL Headers and Libs1077        id: install_opencl1078        if: ${{ matrix.backend == 'opencl-adreno' && matrix.arch == 'arm64' }}1079        run: |1080          git clone https://github.com/KhronosGroup/OpenCL-Headers1081          cd OpenCL-Headers1082          cmake -B build `1083            -DBUILD_TESTING=OFF `1084            -DOPENCL_HEADERS_BUILD_TESTING=OFF `1085            -DOPENCL_HEADERS_BUILD_CXX_TESTS=OFF `1086            -DCMAKE_INSTALL_PREFIX="$env:RUNNER_TEMP/opencl-arm64-release"1087          cmake --build build --target install1088          git clone https://github.com/KhronosGroup/OpenCL-ICD-Loader1089          cd OpenCL-ICD-Loader1090          cmake -B build-arm64-release `1091            -A arm64 `1092            -DCMAKE_PREFIX_PATH="$env:RUNNER_TEMP/opencl-arm64-release" `1093            -DCMAKE_INSTALL_PREFIX="$env:RUNNER_TEMP/opencl-arm64-release"1094          cmake --build build-arm64-release --target install --config release1095 1096      - name: Build1097        id: cmake_build1098        run: |1099          cmake -S . -B build ${{ matrix.defines }} -DGGML_NATIVE=OFF -DGGML_CPU=OFF -DGGML_BACKEND_DL=ON -DLLAMA_BUILD_BORINGSSL=ON1100          cmake --build build --config Release --target ${{ matrix.target }}1101 1102      #- name: ccache-clear1103      #  uses: ./.github/actions/ccache-clear1104      #  with:1105      #    key: release-windows-2025-${{ matrix.arch }}-${{ matrix.backend }}1106 1107      - name: Pack artifacts1108        id: pack_artifacts1109        run: |1110          7z a -snl llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip .\build\bin\Release\${{ matrix.target }}.dll1111 1112      - name: Upload artifacts1113        uses: actions/upload-artifact@v61114        with:1115          path: llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip1116          name: llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip1117 1118  # note: builds only the ggml-cuda backend - llama-server is injected from the1119  #       windows-cpu zip during the release "Merge artifacts" step1120  windows-cuda:1121    name: windows-cuda (${{ matrix.cuda }}, ${{ matrix.arch }})1122    needs: [check-release]1123    if: ${{ needs.check-release.outputs.should_release == 'true' }}1124 1125    runs-on: windows-20221126 1127    permissions:1128      actions: write1129 1130    strategy:1131      matrix:1132        include:1133          - cuda: '12.4'1134            arch: x641135            defines: '-DGGML_CUDA_CUB_3DOT2=ON'1136          - cuda: '13.4'1137            arch: x641138            defines: ''1139          - cuda: '13.4'1140            arch: arm641141            defines: '-DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'1142 1143    steps:1144      - name: Clone1145        id: checkout1146        uses: actions/checkout@v61147 1148      - name: Install Cuda Toolkit1149        uses: ./.github/actions/windows-setup-cuda1150        with:1151          cuda_version: ${{ matrix.cuda }}1152          cuda_arch: ${{ matrix.arch }}1153 1154      - name: Install Ninja1155        id: install_ninja1156        run: |1157          choco install ninja1158 1159      - name: ccache1160        uses: ggml-org/ccache-action@v1.2.241161        with:1162          key: release-windows-2022-${{ matrix.arch }}-cuda-${{ matrix.cuda }}1163          evict-old-files: 1d1164 1165      - name: Build1166        id: cmake_build1167        shell: cmd1168        # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project1169        run: |1170          call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }}1171          cmake -S . -B build -G "Ninja Multi-Config" ^1172            -DGGML_BACKEND_DL=ON ^1173            -DGGML_NATIVE=OFF ^1174            -DGGML_CPU=OFF ^1175            -DGGML_CUDA=ON ^1176            -DLLAMA_BUILD_BORINGSSL=ON ${{ matrix.defines }}1177          set /A NINJA_JOBS=%NUMBER_OF_PROCESSORS%-11178          cmake --build build --config Release -j %NINJA_JOBS% --target ggml-cuda1179 1180      - name: Pack artifacts1181        id: pack_artifacts1182        run: |1183          7z a -snl llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip .\build\bin\Release\ggml-cuda.dll1184 1185      - name: Upload artifacts1186        uses: actions/upload-artifact@v61187        with:1188          path: llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip1189          name: llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip1190 1191      - name: Copy and pack Cuda runtime (x64)1192        if: ${{ matrix.arch == 'x64' }}1193        run: |1194          echo "Cuda install location: ${{ env.CUDA_PATH }}"1195          $dst='.\build\bin\cudart\'1196          robocopy "${{env.CUDA_PATH}}\bin" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll1197          robocopy "${{env.CUDA_PATH}}\lib" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll1198          robocopy "${{env.CUDA_PATH}}\bin\x64" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll1199          7z a cudart-llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip $dst\*1200 

Showing the first 1,200 of 1922 lines. Download the file for the rest.