Felipe97/llama-cpp-compiled
01.2k
1name: Release2 3on:4 workflow_dispatch: # allows manual triggering5 inputs:6 create_release:7 description: 'Create new release'8 required: true9 type: boolean10 push:11 branches:12 - master13 paths: [14 '.github/workflows/release.yml',15 '**/CMakeLists.txt',16 '**/.cmake',17 '**/*.h',18 '**/*.hpp',19 '**/*.c',20 '**/*.cpp',21 '**/*.cu',22 '**/*.cuh',23 '**/*.swift',24 '**/*.m',25 '**/*.metal',26 '**/*.comp',27 '**/*.glsl'28 ]29 30env:31 GH_TOKEN: ${{ github.token }}32 BRANCH_NAME: ${{ github.head_ref || github.ref_name }}33 CMAKE_ARGS: "-DLLAMA_BUILD_EXAMPLES=OFF -DLLAMA_BUILD_TESTS=OFF -DLLAMA_BUILD_TOOLS=ON -DLLAMA_BUILD_SERVER=ON -DGGML_RPC=ON"34 35# note: run this workflow one at a time for better cache reuse36concurrency:37 group: release38 queue: max39 40jobs:41 check-release:42 runs-on: ubuntu-slim43 44 outputs:45 should_release: ${{ steps.check.outputs.should_release }}46 47 steps:48 - id: check49 env:50 COMMIT_MESSAGE: ${{ github.event.head_commit.message }}51 run: |52 if [[ "${{ github.event_name }}" == "workflow_dispatch" ]]; then53 echo "should_release=true" >> $GITHUB_OUTPUT54 elif [[ "${{ github.event_name }}" == "push" && "${{ github.ref }}" == "refs/heads/master" ]]; then55 if echo "$COMMIT_MESSAGE" | grep -q '\[no release\]'; then56 echo "should_release=false" >> $GITHUB_OUTPUT57 else58 echo "should_release=true" >> $GITHUB_OUTPUT59 fi60 else61 echo "should_release=false" >> $GITHUB_OUTPUT62 fi63 64 macos-cpu:65 needs: [check-release, ui-build]66 if: ${{ needs.check-release.outputs.should_release == 'true' }}67 strategy:68 matrix:69 include:70 - build: 'arm64'71 arch: 'arm64'72 os: macos-2673 defines: "-DGGML_METAL_EMBED_LIBRARY=ON -DCMAKE_OSX_DEPLOYMENT_TARGET=13.3"74 # TODO: this build is disabled to save Github Actions resources (https://github.com/ggml-org/llama.cpp/pull/23780)75 # in order to enable it again, we have to provision dedicated runners to run it76 #- build: 'arm64-kleidiai'77 # arch: 'arm64'78 # os: macos-1479 # defines: "-DGGML_METAL_EMBED_LIBRARY=ON -DCMAKE_OSX_DEPLOYMENT_TARGET=13.3 -DGGML_CPU_KLEIDIAI=ON"80 - build: 'x64'81 arch: 'x64'82 os: macos-15-intel83 # Metal is disabled on x64 due to intermittent failures with Github runners not having a GPU:84 # https://github.com/ggml-org/llama.cpp/actions/runs/8635935781/job/23674807267#step:5:231385 defines: "-DGGML_METAL=OFF -DCMAKE_OSX_DEPLOYMENT_TARGET=13.3"86 87 runs-on: ${{ matrix.os }}88 89 permissions:90 actions: write91 92 steps:93 - name: Clone94 id: checkout95 uses: actions/checkout@v696 with:97 fetch-depth: 098 99 - name: Download UI build100 uses: actions/download-artifact@v7101 with:102 name: llama-ui.zip103 path: tools/ui/dist104 105 - name: ccache106 uses: ggml-org/ccache-action@v1.2.24107 with:108 key: release-${{ matrix.os }}-${{ matrix.arch }}109 evict-old-files: 1d110 111 - name: Build112 id: cmake_build113 run: |114 sysctl -a115 cmake -B build \116 ${{ matrix.defines }} \117 -DCMAKE_INSTALL_RPATH='@loader_path' \118 -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \119 -DLLAMA_FATAL_WARNINGS=ON \120 -DLLAMA_BUILD_BORINGSSL=ON \121 ${{ env.CMAKE_ARGS }}122 cmake --build build --config Release -j $(sysctl -n hw.logicalcpu)123 124 - name: Determine tag name125 id: tag126 uses: ./.github/actions/get-tag-name127 128 - name: Pack artifacts129 id: pack_artifacts130 run: |131 cp LICENSE ./build/bin/132 tar -czvf llama-${{ steps.tag.outputs.name }}-bin-macos-${{ matrix.build }}.tar.gz -s ",^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .133 134 - name: Upload artifacts135 uses: actions/upload-artifact@v6136 with:137 path: llama-${{ steps.tag.outputs.name }}-bin-macos-${{ matrix.build }}.tar.gz138 name: llama-bin-macos-${{ matrix.build }}.tar.gz139 140 - name: ccache-clear141 uses: ./.github/actions/ccache-clear142 with:143 key: release-${{ matrix.os }}-${{ matrix.arch }}144 145 ubuntu-cpu:146 needs: [check-release, ui-build]147 if: ${{ needs.check-release.outputs.should_release == 'true' }}148 strategy:149 matrix:150 include:151 - build: 'x64'152 os: ubuntu-22.04153 - build: 'arm64'154 os: ubuntu-24.04-arm155 - build: 's390x'156 os: ubuntu-24.04-s390x157 158 runs-on: ${{ matrix.os }}159 160 permissions:161 actions: write162 163 steps:164 - name: Clone165 id: checkout166 uses: actions/checkout@v6167 with:168 fetch-depth: 0169 170 - name: Download UI build171 uses: actions/download-artifact@v7172 with:173 name: llama-ui.zip174 path: tools/ui/dist175 176 - name: Dependencies177 id: depends178 run: |179 sudo apt-get update180 sudo apt-get install build-essential libssl-dev181 182 - name: Toolchain workaround (GCC 14)183 if: ${{ contains(matrix.os, 'ubuntu-24.04') }}184 run: |185 sudo apt-get install -y gcc-14 g++-14186 echo "CC=gcc-14" >> "$GITHUB_ENV"187 echo "CXX=g++-14" >> "$GITHUB_ENV"188 189 - name: ccache190 if: ${{ matrix.build != 's390x' }}191 uses: ggml-org/ccache-action@v1.2.24192 with:193 key: release-${{ matrix.os }}-cpu194 evict-old-files: 1d195 196 - name: Build197 id: cmake_build198 run: |199 cmake -B build \200 -DCMAKE_INSTALL_RPATH='$ORIGIN' \201 -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \202 -DGGML_BACKEND_DL=ON \203 -DGGML_NATIVE=OFF \204 -DGGML_CPU_ALL_VARIANTS=ON \205 -DLLAMA_FATAL_WARNINGS=ON \206 ${{ env.CMAKE_ARGS }}207 cmake --build build --config Release -j $(nproc)208 209 - name: Determine tag name210 id: tag211 uses: ./.github/actions/get-tag-name212 213 - name: Pack artifacts214 id: pack_artifacts215 run: |216 cp LICENSE ./build/bin/217 tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .218 219 - name: Upload artifacts220 uses: actions/upload-artifact@v6221 with:222 path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-${{ matrix.build }}.tar.gz223 name: llama-bin-ubuntu-${{ matrix.build }}.tar.gz224 225 - name: ccache-clear226 if: ${{ matrix.build != 's390x' }}227 uses: ./.github/actions/ccache-clear228 with:229 key: release-${{ matrix.os }}-cpu230 231 ubuntu-vulkan:232 needs: [check-release, ui-build]233 if: ${{ needs.check-release.outputs.should_release == 'true' }}234 235 strategy:236 matrix:237 include:238 - build: 'x64'239 os: ubuntu-22.04240 - build: 'arm64'241 os: ubuntu-24.04-arm242 243 runs-on: ${{ matrix.os }}244 245 permissions:246 actions: write247 248 steps:249 - name: Clone250 id: checkout251 uses: actions/checkout@v6252 with:253 fetch-depth: 0254 255 - name: Download UI build256 uses: actions/download-artifact@v7257 with:258 name: llama-ui.zip259 path: tools/ui/dist260 261 - name: Dependencies262 id: depends263 run: |264 if [[ "${{ matrix.os }}" =~ "ubuntu-22.04" ]]; then265 wget -qO - https://packages.lunarg.com/lunarg-signing-key-pub.asc | sudo apt-key add -266 sudo wget -qO /etc/apt/sources.list.d/lunarg-vulkan-jammy.list https://packages.lunarg.com/vulkan/lunarg-vulkan-jammy.list267 sudo apt-get update -y268 sudo apt-get install -y build-essential mesa-vulkan-drivers vulkan-sdk libssl-dev269 else270 sudo apt-get update -y271 sudo apt-get install -y gcc-14 g++-14 build-essential glslc libvulkan-dev spirv-headers libssl-dev ninja-build272 echo "CC=gcc-14" >> "$GITHUB_ENV"273 echo "CXX=g++-14" >> "$GITHUB_ENV"274 fi275 276 - name: ccache277 uses: ggml-org/ccache-action@v1.2.24278 with:279 key: release-${{ matrix.os }}-vulkan280 evict-old-files: 1d281 282 - name: Build283 id: cmake_build284 run: |285 cmake -B build \286 -DCMAKE_INSTALL_RPATH='$ORIGIN' \287 -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \288 -DGGML_BACKEND_DL=ON \289 -DGGML_NATIVE=OFF \290 -DGGML_CPU_ALL_VARIANTS=ON \291 -DGGML_VULKAN=ON \292 ${{ env.CMAKE_ARGS }}293 cmake --build build --config Release -j $(nproc)294 295 - name: Determine tag name296 id: tag297 uses: ./.github/actions/get-tag-name298 299 - name: Pack artifacts300 id: pack_artifacts301 run: |302 cp LICENSE ./build/bin/303 tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .304 305 - name: Upload artifacts306 uses: actions/upload-artifact@v6307 with:308 path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz309 name: llama-bin-ubuntu-vulkan-${{ matrix.build }}.tar.gz310 311 - name: ccache-clear312 uses: ./.github/actions/ccache-clear313 with:314 key: release-${{ matrix.os }}-vulkan315 316 ubuntu-cuda:317 name: ubuntu-cuda (${{ matrix.label }}, ${{ matrix.build }})318 needs: [check-release, ui-build]319 if: ${{ needs.check-release.outputs.should_release == 'true' }}320 321 strategy:322 matrix:323 include:324 # label = short version used in artifact names / release body325 # cuda = full container image tag326 - build: 'x64'327 os: ubuntu-24.04328 cuda: '12.8.2'329 label: '12.8'330 defines: '-DGGML_CUDA_CUB_3DOT2=ON'331 - build: 'x64'332 os: ubuntu-24.04333 cuda: '13.3.1'334 label: '13.3'335 defines: ''336 - build: 'arm64'337 os: ubuntu-24.04-arm338 cuda: '13.3.1'339 label: '13.3'340 defines: ''341 342 runs-on: ${{ matrix.os }}343 container: nvidia/cuda:${{ matrix.cuda }}-devel-ubuntu24.04344 345 permissions:346 actions: write347 348 steps:349 # the container has no git; install it before checkout so that a real git350 # repository is created (the get-tag-name action and the build both need it)351 - name: Install git352 run: |353 apt-get update354 apt-get install -y --no-install-recommends git355 356 - name: Clone357 id: checkout358 uses: actions/checkout@v6359 with:360 fetch-depth: 0361 362 # checkout runs as the host user; in-container steps run as root, so git363 # refuses to touch a repo it does not own. Mark the workspace as safe.364 # use the env var: the github.workspace context holds the HOST path,365 # GITHUB_WORKSPACE the container path366 - name: Git safe directory367 run: git config --global --add safe.directory "$GITHUB_WORKSPACE"368 369 - name: Download UI build370 uses: actions/download-artifact@v7371 with:372 name: llama-ui.zip373 path: tools/ui/dist374 375 - name: Dependencies376 id: depends377 # container jobs default to sh (dash); need bash for the [[ ]] below378 shell: bash379 run: |380 apt-get update381 apt-get install -y --no-install-recommends build-essential cmake ninja-build libssl-dev jq python3-venv382 # the container ships GCC 13, which does not know the 'sme' march383 # feature used by the armv9.2 CPU variant of GGML_CPU_ALL_VARIANTS384 if [[ "${{ matrix.build }}" == "arm64" ]]; then385 apt-get install -y --no-install-recommends gcc-14 g++-14386 echo "CC=gcc-14" >> "$GITHUB_ENV"387 echo "CXX=g++-14" >> "$GITHUB_ENV"388 fi389 390 - name: ccache391 uses: ggml-org/ccache-action@v1.2.24392 with:393 key: release-ubuntu-${{ matrix.os }}-cuda-${{ matrix.label }}-${{ matrix.build }}394 evict-old-files: 1d395 max-size: "1G"396 397 - name: Build398 id: cmake_build399 # no CMAKE_CUDA_ARCHITECTURES: use the broad default arch set from400 # ggml/src/ggml-cuda/CMakeLists.txt so the release binary covers many GPUs401 run: |402 cmake -B build \403 -DCMAKE_INSTALL_RPATH='$ORIGIN' \404 -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \405 -DGGML_BACKEND_DL=ON \406 -DGGML_NATIVE=OFF \407 -DGGML_CPU_ALL_VARIANTS=ON \408 -DGGML_CUDA=ON \409 -DGGML_CUDA_NCCL=OFF \410 ${{ env.CMAKE_ARGS }} ${{ matrix.defines }}411 cmake --build build --config Release -j $(nproc)412 413 - name: Determine tag name414 id: tag415 uses: ./.github/actions/get-tag-name416 417 - name: Pack artifacts418 id: pack_artifacts419 run: |420 cp LICENSE ./build/bin/421 tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .422 423 - name: Upload artifacts424 uses: actions/upload-artifact@v6425 with:426 path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz427 name: llama-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz428 429 # ship the CUDA runtime libraries the backend links against, mirroring430 # the windows-cuda cudart zip - extract next to the binaries ($ORIGIN rpath)431 - name: Pack CUDA runtime432 id: pack_cuda_runtime433 run: |434 major="${{ matrix.label }}"435 major="${major%%.*}"436 mkdir -p ./cudart437 # cp -L dereferences the SONAME symlinks into plain files, so the438 # tarball holds exactly 3 files with no versioned duplicates439 cp -L /usr/local/cuda/lib64/libcudart.so.${major} ./cudart/440 cp -L /usr/local/cuda/lib64/libcublas.so.${major} ./cudart/441 cp -L /usr/local/cuda/lib64/libcublasLt.so.${major} ./cudart/442 tar -czvf cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz --transform "s,^\.,cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}," -C ./cudart .443 444 - name: Upload CUDA runtime445 uses: actions/upload-artifact@v6446 with:447 path: cudart-llama-${{ steps.tag.outputs.name }}-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz448 name: cudart-llama-bin-ubuntu-cuda-${{ matrix.label }}-${{ matrix.build }}.tar.gz449 450 - name: ccache-clear451 uses: ./.github/actions/ccache-clear452 with:453 key: release-ubuntu-${{ matrix.os }}-cuda-${{ matrix.label }}-${{ matrix.build }}454 455 android-arm64:456 needs: [check-release, ui-build]457 if: ${{ needs.check-release.outputs.should_release == 'true' }}458 459 runs-on: ubuntu-24.04 # previously ubuntu-latest460 461 #permissions:462 # actions: write463 464 env:465 NDK_VERSION: "29.0.14206865"466 467 steps:468 - name: Clone469 id: checkout470 uses: actions/checkout@v6471 with:472 fetch-depth: 0473 474 - name: Download UI build475 uses: actions/download-artifact@v7476 with:477 name: llama-ui.zip478 path: tools/ui/dist479 480 - name: Set up JDK481 uses: actions/setup-java@v5482 with:483 java-version: 17484 distribution: temurin485 486 - name: Setup Android SDK487 uses: android-actions/setup-android@be39fa834029ff78f1a44aa3bb0819b8fc2bd8fd # v4.0.4488 with:489 log-accepted-android-sdk-licenses: false490 491 - name: Install NDK492 run: |493 sdkmanager "ndk;${{ env.NDK_VERSION }}"494 echo "ANDROID_NDK=${ANDROID_SDK_ROOT}/ndk/${{ env.NDK_VERSION }}" >> $GITHUB_ENV495 496 # note : disabled to spare some cache space (https://github.com/ggml-org/llama.cpp/pull/23789)497 # for some reason, the ccache does not improve the build time in this case498 # example:499 # cache off: https://github.com/ggerganov/tmp2/actions/runs/26534713799/job/78160400831500 # cache on: https://github.com/ggerganov/tmp2/actions/runs/26534713799/job/78224189394501 #502 #- name: ccache503 # uses: ggml-org/ccache-action@v1.2.24504 # with:505 # key: release-android-arm64506 # evict-old-files: 1d507 508 - name: Build509 id: cmake_build510 run: |511 cmake -B build \512 -DCMAKE_TOOLCHAIN_FILE=${ANDROID_NDK}/build/cmake/android.toolchain.cmake \513 -DANDROID_ABI=arm64-v8a \514 -DANDROID_PLATFORM=android-28 \515 -DCMAKE_INSTALL_RPATH='$ORIGIN' \516 -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \517 -DGGML_BACKEND_DL=ON \518 -DGGML_NATIVE=OFF \519 -DGGML_CPU_ALL_VARIANTS=ON \520 -DLLAMA_FATAL_WARNINGS=ON \521 -DGGML_OPENMP=OFF \522 -DLLAMA_BUILD_BORINGSSL=ON \523 ${{ env.CMAKE_ARGS }}524 cmake --build build --config Release -j $(nproc)525 526 #- name: ccache-clear527 # uses: ./.github/actions/ccache-clear528 # with:529 # key: release-android-arm64530 531 - name: Determine tag name532 id: tag533 uses: ./.github/actions/get-tag-name534 535 - name: Pack artifacts536 id: pack_artifacts537 run: |538 cp LICENSE ./build/bin/539 tar -czvf llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C ./build/bin .540 541 - name: Upload artifacts542 uses: actions/upload-artifact@v6543 with:544 path: llama-${{ steps.tag.outputs.name }}-bin-android-arm64.tar.gz545 name: llama-bin-android-arm64.tar.gz546 547 ubuntu-24-openvino:548 needs: [check-release, ui-build]549 if: ${{ needs.check-release.outputs.should_release == 'true' }}550 551 runs-on: ubuntu-24.04552 553 permissions:554 actions: write555 556 outputs:557 openvino_version: ${{ steps.openvino_version.outputs.value }}558 559 env:560 # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile561 OPENVINO_VERSION_MAJOR: "2026.4"562 OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"563 564 steps:565 - name: Set OpenVINO version output566 id: openvino_version567 run: echo "value=${{ env.OPENVINO_VERSION_MAJOR }}" >> $GITHUB_OUTPUT568 569 - name: Clone570 id: checkout571 uses: actions/checkout@v6572 with:573 fetch-depth: 0574 575 - name: Download UI build576 uses: actions/download-artifact@v7577 with:578 name: llama-ui.zip579 path: tools/ui/dist580 581 - name: ccache582 uses: ggml-org/ccache-action@v1.2.24583 with:584 key: release-ubuntu-24.04-openvino-release-no-preset-v1585 evict-old-files: 1d586 587 - name: Dependencies588 run: |589 sudo apt-get update590 sudo apt-get install -y build-essential libssl-dev libtbb12 cmake ninja-build python3-pip591 sudo apt install ocl-icd-opencl-dev opencl-headers opencl-clhpp-headers intel-opencl-icd592 593 - name: Use OpenVINO Toolkit Cache594 uses: actions/cache@v5595 id: cache-openvino596 with:597 path: ./openvino_toolkit598 key: cache-gha-openvino-toolkit-v${{ env.OPENVINO_VERSION_FULL }}-${{ runner.os }}599 600 - name: Setup OpenVINO Toolkit601 if: steps.cache-openvino.outputs.cache-hit != 'true'602 uses: ./.github/actions/linux-setup-openvino603 with:604 path: ./openvino_toolkit605 version_major: ${{ env.OPENVINO_VERSION_MAJOR }}606 version_full: ${{ env.OPENVINO_VERSION_FULL }}607 608 - name: Install OpenVINO dependencies609 run: |610 cd ./openvino_toolkit611 chmod +x ./install_dependencies/install_openvino_dependencies.sh612 echo "Y" | sudo -E ./install_dependencies/install_openvino_dependencies.sh613 614 - name: Build615 id: cmake_build616 run: |617 source ./openvino_toolkit/setupvars.sh618 cmake -B build/ReleaseOV -G Ninja \619 -DCMAKE_BUILD_TYPE=Release \620 -DGGML_OPENVINO=ON \621 -DCMAKE_INSTALL_RPATH='$ORIGIN' \622 -DCMAKE_BUILD_WITH_INSTALL_RPATH=ON \623 ${{ env.CMAKE_ARGS }}624 cmake --build build/ReleaseOV --config Release --parallel625 626 - name: Determine tag name627 id: tag628 uses: ./.github/actions/get-tag-name629 630 - name: Pack artifacts631 id: pack_artifacts632 run: |633 dest=./build/ReleaseOV/bin634 OPENVINO_ROOT=./openvino_toolkit635 ov_lib="$OPENVINO_ROOT/runtime/lib/intel64"636 637 # Bundle OpenVINO runtime libs + TBB. Binaries built with RPATH=$ORIGIN638 # load these siblings without setupvars.sh / LD_LIBRARY_PATH.639 cp -P "$ov_lib"/libopenvino.so* \640 "$ov_lib"/libopenvino_c.so* \641 "$ov_lib"/libopenvino_*_plugin.so \642 "$ov_lib"/libopenvino_intel_npu_compiler*.so \643 "$OPENVINO_ROOT"/runtime/3rdparty/tbb/lib/*.so* \644 "$dest"645 cp -P /usr/lib/x86_64-linux-gnu/libOpenCL.so.1* "$dest" 2>/dev/null || true646 cp "$ov_lib"/cache.json "$dest" 2>/dev/null || true647 648 # OpenVINO licensing649 cp -r "$OPENVINO_ROOT"/docs/licensing "$dest"/openvino-licensing650 651 cp LICENSE "$dest"652 tar -czvf llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz --transform "s,^\.,llama-${{ steps.tag.outputs.name }}," -C "$dest" .653 654 - name: Upload artifacts655 uses: actions/upload-artifact@v6656 with:657 path: llama-${{ steps.tag.outputs.name }}-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz658 name: llama-bin-ubuntu-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.tar.gz659 660 - name: ccache-clear661 uses: ./.github/actions/ccache-clear662 with:663 key: release-ubuntu-24.04-openvino-release-no-preset-v1664 665 windows-openvino:666 needs: [check-release, ui-build]667 if: ${{ needs.check-release.outputs.should_release == 'true' }}668 669 runs-on: windows-2022670 671 outputs:672 openvino_version: ${{ steps.openvino_version.outputs.value }}673 674 env:675 # Sync versions in build-openvino.yml, build-self-hosted.yml, release.yml, build-cache.yml, .devops/openvino.Dockerfile676 OPENVINO_VERSION_MAJOR: "2026.4"677 OPENVINO_VERSION_FULL: "2026.4.0.22959.99c81491cc3"678 679 steps:680 - name: Set OpenVINO version output681 id: openvino_version682 shell: bash683 run: echo "value=${{ env.OPENVINO_VERSION_MAJOR }}" >> $GITHUB_OUTPUT684 685 - name: Clone686 id: checkout687 uses: actions/checkout@v6688 with:689 fetch-depth: 0690 691 - name: Download UI build692 uses: actions/download-artifact@v7693 with:694 name: llama-ui.zip695 path: tools/ui/dist696 697 - name: ccache698 uses: ggml-org/ccache-action@v1.2.24699 with:700 key: release-windows-2022-openvino701 variant: ccache702 evict-old-files: 1d703 704 - name: Setup Cache705 uses: actions/cache@v5706 id: cache-openvino707 with:708 path: ./openvino_toolkit709 key: cache-gha-openvino-toolkit-v${{ env.OPENVINO_VERSION_FULL }}-${{ runner.os }}710 711 - name: Setup OpenVINO Toolkit712 if: steps.cache-openvino.outputs.cache-hit != 'true'713 uses: ./.github/actions/windows-setup-openvino714 with:715 path: ./openvino_toolkit716 version_major: ${{ env.OPENVINO_VERSION_MAJOR }}717 version_full: ${{ env.OPENVINO_VERSION_FULL }}718 719 - name: Install OpenCL using vcpkg720 shell: powershell721 run: |722 git clone https://github.com/microsoft/vcpkg C:\vcpkg723 C:\vcpkg\bootstrap-vcpkg.bat724 C:\vcpkg\vcpkg install opencl725 726 - name: Build727 id: cmake_build728 shell: cmd729 run: |730 REM Find extracted OpenVINO folder dynamically731 for /d %%i in (openvino_toolkit\*) do set OPENVINO_ROOT=%%i732 733 if not exist "%OPENVINO_ROOT%\runtime\cmake\OpenVINOConfig.cmake" (734 echo ERROR: OpenVINOConfig.cmake not found735 exit /b 1736 )737 738 call "%OPENVINO_ROOT%\setupvars.bat"739 740 cmake -B build\ReleaseOV -G "Visual Studio 17 2022" ^741 -A x64 ^742 -DCMAKE_BUILD_TYPE=Release ^743 -DGGML_OPENVINO=ON ^744 -DLLAMA_BUILD_BORINGSSL=ON ^745 -DCMAKE_TOOLCHAIN_FILE=C:\vcpkg\scripts\buildsystems\vcpkg.cmake ^746 ${{ env.CMAKE_ARGS }}747 748 cmake --build build\ReleaseOV --config Release -- /m749 750 - name: Determine tag name751 id: tag752 uses: ./.github/actions/get-tag-name753 754 - name: Pack artifacts755 id: pack_artifacts756 shell: powershell757 run: |758 # Locate the extracted OpenVINO toolkit root (same pattern as the Build step).759 $OPENVINO_ROOT = (Get-ChildItem -Directory openvino_toolkit | Select-Object -First 1).FullName760 if (-not $OPENVINO_ROOT) {761 Write-Error "OpenVINO toolkit folder not found under .\openvino_toolkit"762 exit 1763 }764 765 $dest = ".\build\ReleaseOV\bin\Release"766 767 $ovBin = Join-Path $OPENVINO_ROOT 'runtime\bin\intel64\Release'768 Copy-Item -Path (Join-Path $ovBin '*.dll') -Destination $dest -Force769 Copy-Item -Path (Join-Path $ovBin 'cache.json') -Destination $dest -Force770 771 $tbbBin = Join-Path $OPENVINO_ROOT 'runtime\3rdparty\tbb\bin'772 Copy-Item -Path (Join-Path $tbbBin 'tbb*.dll') -Destination $dest -Force773 774 # OpenVINO licensing775 $licensingDest = Join-Path $dest 'openvino-licensing'776 New-Item -ItemType Directory -Force -Path $licensingDest | Out-Null777 Copy-Item -Path (Join-Path $OPENVINO_ROOT 'docs\licensing\*') -Destination $licensingDest -Recurse -Force778 779 Copy-Item LICENSE $dest780 7z a -snl llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip $dest\*781 782 - name: Upload artifacts783 uses: actions/upload-artifact@v6784 with:785 path: llama-${{ steps.tag.outputs.name }}-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip786 name: llama-bin-win-openvino-${{ env.OPENVINO_VERSION_MAJOR }}-x64.zip787 788 - name: ccache-clear789 uses: ./.github/actions/ccache-clear790 with:791 key: release-windows-2022-openvino792 793 windows-cpu:794 name: windows-cpu / ${{ matrix.arch }}795 needs: [check-release, ui-build]796 if: ${{ needs.check-release.outputs.should_release == 'true' }}797 798 runs-on: windows-2025-vs2026799 800 permissions:801 actions: write802 803 strategy:804 matrix:805 include:806 - arch: 'x64'807 - arch: 'arm64'808 809 steps:810 - name: Clone811 uses: actions/checkout@v6812 with:813 fetch-depth: 0814 815 - name: Download UI build816 uses: actions/download-artifact@v7817 with:818 name: llama-ui.zip819 path: tools/ui/dist820 821 - name: Install Ninja822 run: |823 choco install ninja824 825 - name: ccache826 uses: ggml-org/ccache-action@v1.2.24827 with:828 key: release-windows-2025-vs2026-${{ matrix.arch }}-cpu829 evict-old-files: 1d830 831 - name: Build832 shell: cmd833 run: |834 call "C:\Program Files\Microsoft Visual Studio\18\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }}835 cmake -S . -B build -G "Ninja Multi-Config" ^836 -D CMAKE_TOOLCHAIN_FILE=cmake/${{ matrix.arch }}-windows-llvm.cmake ^837 -DLLAMA_BUILD_BORINGSSL=ON ^838 -DGGML_NATIVE=OFF ^839 -DGGML_BACKEND_DL=ON ^840 -DGGML_CPU_ALL_VARIANTS=${{ matrix.arch == 'x64' && 'ON' || 'OFF' }} ^841 -DGGML_OPENMP=ON ^842 -DGGML_OPENMP_FETCH=ON ^843 ${{ env.CMAKE_ARGS }}844 cmake --build build --config Release845 846 - name: Pack artifacts847 id: pack_artifacts848 run: |849 7z a -snl llama-bin-win-cpu-${{ matrix.arch }}.zip .\build\bin\Release\*850 851 - name: Upload artifacts852 uses: actions/upload-artifact@v6853 with:854 path: llama-bin-win-cpu-${{ matrix.arch }}.zip855 name: llama-bin-win-cpu-${{ matrix.arch }}.zip856 857 - name: ccache-clear858 uses: ./.github/actions/ccache-clear859 with:860 key: release-windows-2025-vs2026-${{ matrix.arch }}-cpu861 862 # note: builds only the ggml-hip backend - llama-server is injected from the863 # windows-cpu zip during the release "Merge artifacts" step864 windows-rocm:865 needs: [check-release]866 if: ${{ needs.check-release.outputs.should_release == 'true' }}867 868 runs-on: windows-2022869 870 strategy:871 matrix:872 include:873 - ROCM_VERSION: "10.0.0"874 gpu_targets: "gfx1010;gfx1011;gfx1012;gfx1030;gfx1031;gfx1032;gfx1033;gfx1034;gfx1035;gfx1036;gfx1100;gfx1101;gfx1102;gfx1103;gfx1150;gfx1151;gfx1152;gfx1153;gfx1200;gfx1201"875 build: x64876 877 steps:878 - name: Clone879 id: checkout880 uses: actions/checkout@v6881 with:882 fetch-depth: 0883 884 - name: Install Ninja885 run: |886 choco install ninja887 888 - name: ccache889 uses: ggml-org/ccache-action@v1.2.24890 with:891 key: windows-rocm-${{ matrix.ROCM_VERSION }}-${{ matrix.build }}892 evict-old-files: 1d893 max-size: "1G"894 895 # - name: Cache ROCm Installation896 # id: cache-rocm897 # uses: actions/cache@v5898 # with:899 # path: C:\TheRock\build900 # key: rocm-wheels-${{ matrix.ROCM_VERSION }}-multi-arch-${{ runner.os }}901 902 - name: Setup ROCm903 # if: steps.cache-rocm.outputs.cache-hit != 'true'904 uses: ./.github/actions/windows-setup-rocm905 with:906 version: ${{ matrix.ROCM_VERSION }}907 908 - name: Setup ROCm Environment909 run: |910 $ErrorActionPreference = "Stop"911 912 # Activate venv from cache or fresh install913 & C:\TheRock\build\.venv\Scripts\Activate.ps1914 915 # Expand the devel tree (idempotent; no-op if already done during install)916 rocm-sdk init917 if ($LASTEXITCODE -ne 0) { throw "rocm-sdk init failed with exit code $LASTEXITCODE" }918 919 # Get ROCm installation paths using the rocm-sdk CLI tool920 $rocmPath = (rocm-sdk path --root)921 if (-not $rocmPath) { throw "rocm-sdk path --root returned empty - devel package may not be installed" }922 $rocmPath = $rocmPath.Trim()923 $cmakePath = (rocm-sdk path --cmake).Trim()924 $binPath = (rocm-sdk path --bin).Trim()925 write-host "ROCm root: $rocmPath"926 write-host "CMake path: $cmakePath"927 write-host "Bin path: $binPath"928 929 echo "HIP_PATH=$rocmPath" >> $env:GITHUB_ENV930 echo "CMAKE_PREFIX_PATH=$cmakePath" >> $env:GITHUB_ENV931 echo "HIP_DEVICE_LIB_PATH=$rocmPath\lib\llvm\amdgcn\bitcode" >> $env:GITHUB_ENV932 echo "HIP_PLATFORM=amd" >> $env:GITHUB_ENV933 echo "LLVM_PATH=$rocmPath\lib\llvm" >> $env:GITHUB_ENV934 echo "$binPath" >> $env:GITHUB_PATH935 936 # Keep venv in PATH for subsequent steps937 echo "C:\TheRock\build\.venv\Scripts" >> $env:GITHUB_PATH938 939 - name: Build940 run: |941 cmake -S . -B build `942 -G "Ninja Multi-Config" `943 -DCMAKE_PREFIX_PATH="${env:HIP_PATH}" `944 -DGGML_BACKEND_DL=ON `945 -DGGML_NATIVE=OFF `946 -DGGML_CPU=OFF `947 -DGGML_HIP=ON `948 -DCMAKE_C_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang.exe" `949 -DCMAKE_CXX_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang++.exe" `950 -DCMAKE_C_FLAGS="-Wno-error=incompatible-pointer-types" `951 -DCMAKE_HIP_COMPILER="${env:HIP_PATH}\lib\llvm\bin\clang.exe" `952 -DHIP_PATH="${env:HIP_PATH}" `953 -DAMDGPU_TARGETS="${{ matrix.gpu_targets }}"954 cmake --build build --config Release --parallel ${env:NUMBER_OF_PROCESSORS} --target ggml-hip955 956 - name: Verify HIP backend was built957 run: |958 $hipDll = Get-ChildItem -Path build\bin\Release -Filter "ggml-hip*.dll" -ErrorAction SilentlyContinue959 if (-not $hipDll) {960 Write-Host "##[error]ggml-hip*.dll was NOT produced. The HIP backend silently failed to build."961 Write-Host "Contents of build\bin\Release:"962 Get-ChildItem build\bin\Release | Format-Table -AutoSize963 exit 1964 }965 Write-Host "HIP backend artifact found:"966 $hipDll | Format-Table FullName, Length -AutoSize967 968 - name: Determine tag name969 id: tag970 uses: ./.github/actions/get-tag-name971 972 - name: Get ROCm short version973 run: |974 $rocmVersionShort = ('${{ matrix.ROCM_VERSION }}'.Split('.')[0..1] -join '.')975 echo "ROCM_VERSION_SHORT=$rocmVersionShort" >> $env:GITHUB_ENV976 977 - name: Bundle HIP runtime DLLs (amdhip64_7.dll, rocm_kpack.dll, amd_comgr.dll)978 run: |979 $ErrorActionPreference = "Stop"980 # See issue https://github.com/ggml-org/llama.cpp/issues/26929.981 # ggml-hip.dll loads amdhip64_7.dll at run time. The Adrenalin driver982 # ships an amdhip64_7.dll in System32, which the loader searches before PATH,983 # so a matching DLL from PATH cannot win. Copy amdhip64 next to the984 # binaries (exe directory is searched before System32) so the correct985 # runtime is used. rocm_kpack.dll is amdhip64_7's direct dependency, so986 # copy the matching version too. amd_comgr is copied as well to keep it987 # in sync with the bundled amdhip64, avoiding a version mismatch with a988 # amd_comgr from System32.989 # rocblas/hipblaslt kernels resolve fine via PATH and are not copied.990 $binPath = (rocm-sdk path --bin).Trim()991 if (-not $binPath) { throw "rocm-sdk path --bin returned empty" }992 write-host "ROCm bin path: $binPath"993 994 $patterns = @("amdhip64_7.dll", "rocm_kpack.dll", "amd_comgr.dll")995 foreach ($pattern in $patterns) {996 $files = Get-ChildItem -Path $binPath -Filter $pattern -ErrorAction SilentlyContinue997 if (-not $files) { throw "no match for $pattern in $binPath" }998 foreach ($f in $files) {999 Copy-Item $f.FullName -Destination build\bin\Release -Force1000 write-host " copied $($f.Name)"1001 }1002 }1003 1004 - name: Pack artifacts1005 run: |1006 7z a -snl llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip `1007 .\build\bin\Release\ggml-hip.dll `1008 .\build\bin\Release\amdhip64_7.dll `1009 .\build\bin\Release\rocm_kpack.dll `1010 .\build\bin\Release\amd_comgr.dll1011 1012 - name: Upload artifacts1013 uses: actions/upload-artifact@v61014 with:1015 path: llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip1016 name: llama-bin-win-rocm-${{ env.ROCM_VERSION_SHORT }}-${{ matrix.build }}.zip1017 1018 - name: ccache-clear1019 uses: ./.github/actions/ccache-clear1020 with:1021 key: windows-rocm-${{ matrix.ROCM_VERSION }}-${{ matrix.build }}1022 1023 # note: builds only the backend library - llama-server (with the embedded UI)1024 # is injected from the windows-cpu zip during the release "Merge artifacts" step1025 windows:1026 needs: [check-release]1027 if: ${{ needs.check-release.outputs.should_release == 'true' }}1028 1029 runs-on: windows-20251030 1031 permissions:1032 actions: write1033 1034 env:1035 OPENBLAS_VERSION: 0.3.231036 VULKAN_VERSION: 1.4.357.01037 1038 strategy:1039 matrix:1040 include:1041 - backend: 'vulkan'1042 arch: 'x64'1043 defines: '-DGGML_VULKAN=ON'1044 target: 'ggml-vulkan'1045 - backend: 'opencl-adreno'1046 arch: 'arm64'1047 defines: '-G "Ninja Multi-Config" -D CMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-llvm.cmake -DCMAKE_PREFIX_PATH="$env:RUNNER_TEMP/opencl-arm64-release" -DGGML_OPENCL=ON -DGGML_OPENCL_USE_ADRENO_KERNELS=ON'1048 target: 'ggml-opencl'1049 1050 steps:1051 - name: Clone1052 id: checkout1053 uses: actions/checkout@v61054 1055 - name: Install Vulkan SDK1056 id: get_vulkan1057 if: ${{ matrix.backend == 'vulkan' }}1058 run: |1059 curl.exe -o $env:RUNNER_TEMP/VulkanSDK-Installer.exe -L "https://sdk.lunarg.com/sdk/download/${env:VULKAN_VERSION}/windows/vulkansdk-windows-X64-${env:VULKAN_VERSION}.exe"1060 & "$env:RUNNER_TEMP\VulkanSDK-Installer.exe" --accept-licenses --default-answer --confirm-command install1061 Add-Content $env:GITHUB_ENV "VULKAN_SDK=C:\VulkanSDK\${env:VULKAN_VERSION}"1062 Add-Content $env:GITHUB_PATH "C:\VulkanSDK\${env:VULKAN_VERSION}\bin"1063 1064 - name: Install Ninja1065 id: install_ninja1066 run: |1067 choco install ninja1068 1069 # TODO: these jobs need to use llvm toolchain in order to utilize the ccache1070 #- name: ccache1071 # uses: ggml-org/ccache-action@v1.2.241072 # with:1073 # key: release-windows-2025-${{ matrix.arch }}-${{ matrix.backend }}1074 # evict-old-files: 1d1075 1076 - name: Install OpenCL Headers and Libs1077 id: install_opencl1078 if: ${{ matrix.backend == 'opencl-adreno' && matrix.arch == 'arm64' }}1079 run: |1080 git clone https://github.com/KhronosGroup/OpenCL-Headers1081 cd OpenCL-Headers1082 cmake -B build `1083 -DBUILD_TESTING=OFF `1084 -DOPENCL_HEADERS_BUILD_TESTING=OFF `1085 -DOPENCL_HEADERS_BUILD_CXX_TESTS=OFF `1086 -DCMAKE_INSTALL_PREFIX="$env:RUNNER_TEMP/opencl-arm64-release"1087 cmake --build build --target install1088 git clone https://github.com/KhronosGroup/OpenCL-ICD-Loader1089 cd OpenCL-ICD-Loader1090 cmake -B build-arm64-release `1091 -A arm64 `1092 -DCMAKE_PREFIX_PATH="$env:RUNNER_TEMP/opencl-arm64-release" `1093 -DCMAKE_INSTALL_PREFIX="$env:RUNNER_TEMP/opencl-arm64-release"1094 cmake --build build-arm64-release --target install --config release1095 1096 - name: Build1097 id: cmake_build1098 run: |1099 cmake -S . -B build ${{ matrix.defines }} -DGGML_NATIVE=OFF -DGGML_CPU=OFF -DGGML_BACKEND_DL=ON -DLLAMA_BUILD_BORINGSSL=ON1100 cmake --build build --config Release --target ${{ matrix.target }}1101 1102 #- name: ccache-clear1103 # uses: ./.github/actions/ccache-clear1104 # with:1105 # key: release-windows-2025-${{ matrix.arch }}-${{ matrix.backend }}1106 1107 - name: Pack artifacts1108 id: pack_artifacts1109 run: |1110 7z a -snl llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip .\build\bin\Release\${{ matrix.target }}.dll1111 1112 - name: Upload artifacts1113 uses: actions/upload-artifact@v61114 with:1115 path: llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip1116 name: llama-bin-win-${{ matrix.backend }}-${{ matrix.arch }}.zip1117 1118 # note: builds only the ggml-cuda backend - llama-server is injected from the1119 # windows-cpu zip during the release "Merge artifacts" step1120 windows-cuda:1121 name: windows-cuda (${{ matrix.cuda }}, ${{ matrix.arch }})1122 needs: [check-release]1123 if: ${{ needs.check-release.outputs.should_release == 'true' }}1124 1125 runs-on: windows-20221126 1127 permissions:1128 actions: write1129 1130 strategy:1131 matrix:1132 include:1133 - cuda: '12.4'1134 arch: x641135 defines: '-DGGML_CUDA_CUB_3DOT2=ON'1136 - cuda: '13.4'1137 arch: x641138 defines: ''1139 - cuda: '13.4'1140 arch: arm641141 defines: '-DCMAKE_TOOLCHAIN_FILE=cmake/arm64-windows-msvc-cuda.cmake'1142 1143 steps:1144 - name: Clone1145 id: checkout1146 uses: actions/checkout@v61147 1148 - name: Install Cuda Toolkit1149 uses: ./.github/actions/windows-setup-cuda1150 with:1151 cuda_version: ${{ matrix.cuda }}1152 cuda_arch: ${{ matrix.arch }}1153 1154 - name: Install Ninja1155 id: install_ninja1156 run: |1157 choco install ninja1158 1159 - name: ccache1160 uses: ggml-org/ccache-action@v1.2.241161 with:1162 key: release-windows-2022-${{ matrix.arch }}-cuda-${{ matrix.cuda }}1163 evict-old-files: 1d1164 1165 - name: Build1166 id: cmake_build1167 shell: cmd1168 # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project1169 run: |1170 call "C:\Program Files\Microsoft Visual Studio\2022\Enterprise\VC\Auxiliary\Build\vcvarsall.bat" ${{ matrix.arch == 'x64' && 'x64' || 'amd64_arm64' }}1171 cmake -S . -B build -G "Ninja Multi-Config" ^1172 -DGGML_BACKEND_DL=ON ^1173 -DGGML_NATIVE=OFF ^1174 -DGGML_CPU=OFF ^1175 -DGGML_CUDA=ON ^1176 -DLLAMA_BUILD_BORINGSSL=ON ${{ matrix.defines }}1177 set /A NINJA_JOBS=%NUMBER_OF_PROCESSORS%-11178 cmake --build build --config Release -j %NINJA_JOBS% --target ggml-cuda1179 1180 - name: Pack artifacts1181 id: pack_artifacts1182 run: |1183 7z a -snl llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip .\build\bin\Release\ggml-cuda.dll1184 1185 - name: Upload artifacts1186 uses: actions/upload-artifact@v61187 with:1188 path: llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip1189 name: llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip1190 1191 - name: Copy and pack Cuda runtime (x64)1192 if: ${{ matrix.arch == 'x64' }}1193 run: |1194 echo "Cuda install location: ${{ env.CUDA_PATH }}"1195 $dst='.\build\bin\cudart\'1196 robocopy "${{env.CUDA_PATH}}\bin" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll1197 robocopy "${{env.CUDA_PATH}}\lib" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll1198 robocopy "${{env.CUDA_PATH}}\bin\x64" $dst cudart64_*.dll cublas64_*.dll cublasLt64_*.dll1199 7z a cudart-llama-bin-win-cuda-${{ matrix.cuda }}-${{ matrix.arch }}.zip $dst\*1200 