Felipe97/llama-cpp-compiled
01.2k
1name: Server (self-hosted)2 3on:4 workflow_dispatch: # allows manual triggering5 inputs:6 sha:7 description: 'Commit SHA1 to build'8 required: false9 type: string10 slow_tests:11 description: 'Run slow tests'12 required: true13 type: boolean14 push:15 branches:16 - master17 paths: [18 '.github/workflows/server-self-hosted.yml',19 '**/CMakeLists.txt',20 '**/Makefile',21 '**/*.h',22 '**/*.hpp',23 '**/*.c',24 '**/*.cpp',25 '**/*.cu',26 '**/*.swift',27 '**/*.m',28 'tools/server/**.*'29 ]30 31env:32 # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)33 HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}34 LLAMA_ARG_LOG_COLORS: 135 LLAMA_ARG_LOG_PREFIX: 136 LLAMA_ARG_LOG_TIMESTAMPS: 137 LLAMA_ARG_LOG_VERBOSITY: 1038 39concurrency:40 group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}41 cancel-in-progress: true42 43jobs:44 server-metal:45 runs-on: [self-hosted, llama-server, macOS, ARM64]46 47 steps:48 - name: Clone49 id: checkout50 uses: actions/checkout@v651 with:52 fetch-depth: 053 ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}54 55 - name: Build56 id: cmake_build57 run: |58 cmake -B build -DGGML_SCHED_NO_REALLOC=ON59 cmake --build build --config Release -j $(sysctl -n hw.logicalcpu) --target llama-server60 61 - name: Python setup62 id: setup_python63 run: |64 cd tools/server/tests65 python3 -m venv venv66 source venv/bin/activate67 pip install -r requirements.txt68 69 - name: Tests (GPUx1)70 id: server_integration_tests71 if: ${{ !github.event.pull_request }}72 run: |73 cd tools/server/tests74 source venv/bin/activate75 PYTEST_WORKERS=1 ./tests.sh76 77 - name: Tests (GPUx1, backend-sampling)78 id: server_integration_tests_backend_sampling79 if: ${{ !github.event.pull_request }}80 run: |81 cd tools/server/tests82 source venv/bin/activate83 export LLAMA_ARG_BACKEND_SAMPLING=184 PYTEST_WORKERS=1 ./tests.sh85 86 - name: Tests (GPUx2)87 id: server_integration_tests_gpu288 if: ${{ !github.event.pull_request }}89 run: |90 cd tools/server/tests91 source venv/bin/activate92 export GGML_METAL_DEVICES=293 PYTEST_WORKERS=1 ./tests.sh94 95 - name: Tests (GPUx2, backend-sampling)96 id: server_integration_tests_gpu2_backend_sampling97 if: ${{ !github.event.pull_request }}98 run: |99 cd tools/server/tests100 source venv/bin/activate101 export GGML_METAL_DEVICES=2 LLAMA_ARG_BACKEND_SAMPLING=1102 PYTEST_WORKERS=1 ./tests.sh103 104 server-cuda:105 runs-on: "hf-jobs-t4-small:cuda13"106 107 steps:108 - name: Clone109 id: checkout110 uses: actions/checkout@v6111 with:112 fetch-depth: 0113 ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}114 115 - name: Install dependencies116 run: |117 sudo apt update118 sudo apt install -y cmake libssl-dev python3 python3-venv python3-pip119 120 - name: ccache121 uses: ggml-org/ccache-action@v1.2.24122 with:123 restore: false124 save: false125 126 - name: ccache-buckets-restore127 uses: ./.github/actions/ccache-buckets128 with:129 key: self-hosted-server-cuda130 folder: llama.cpp131 hf_bucket: ggml-org/cache132 133 - name: Build134 id: cmake_build135 run: |136 cmake -B build -DGGML_CUDA=ON -DGGML_SCHED_NO_REALLOC=ON -DCMAKE_CUDA_COMPILER=/usr/local/cuda/bin/nvcc137 cmake --build build --config Release -j $(nproc) --target llama-server138 139 - name: ccache-buckets-save140 if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}141 uses: ./.github/actions/ccache-buckets142 env:143 HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}144 with:145 key: self-hosted-server-cuda146 folder: llama.cpp147 evict-old-files: 1d148 hf_bucket: ggml-org/cache149 save: true150 151 - name: Python setup152 id: setup_python153 run: |154 cd tools/server/tests155 python3 -m venv venv156 source venv/bin/activate157 pip install -r requirements.txt158 159 - name: Tests (GPUx1)160 id: server_integration_tests161 if: ${{ !github.event.pull_request }}162 run: |163 cd tools/server/tests164 source venv/bin/activate165 PYTEST_WORKERS=1 ./tests.sh166 167 - name: Tests (GPUx1, backend-sampling)168 id: server_integration_tests_backend_sampling169 if: ${{ !github.event.pull_request }}170 run: |171 cd tools/server/tests172 source venv/bin/activate173 export LLAMA_ARG_BACKEND_SAMPLING=1174 PYTEST_WORKERS=1 ./tests.sh175 176 - name: Tests (GPUx2)177 id: server_integration_tests_gpu2178 if: ${{ !github.event.pull_request }}179 run: |180 cd tools/server/tests181 source venv/bin/activate182 export GGML_CUDA_DEVICES=2183 PYTEST_WORKERS=1 ./tests.sh184 185 - name: Tests (GPUx2, backend-sampling)186 id: server_integration_tests_gpu2_backend_sampling187 if: ${{ !github.event.pull_request }}188 run: |189 cd tools/server/tests190 source venv/bin/activate191 export GGML_CUDA_DEVICES=2 LLAMA_ARG_BACKEND_SAMPLING=1192 PYTEST_WORKERS=1 ./tests.sh193 194 server-kleidiai:195 runs-on: ah-ubuntu_24_04-c8g_8x196 197 steps:198 - name: Clone199 id: checkout200 uses: actions/checkout@v6201 with:202 fetch-depth: 0203 ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}204 205 - name: Dependencies206 id: depends207 run: |208 set -euxo pipefail209 sudo apt-get update210 sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \211 apt-get install -y \212 build-essential \213 libssl-dev \214 python3-venv \215 gpg \216 wget \217 time \218 git-lfs219 220 git lfs install221 222 # install the latest cmake223 sudo install -d /usr/share/keyrings224 wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \225 | gpg --dearmor \226 | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null227 echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \228 | sudo tee /etc/apt/sources.list.d/kitware.list229 sudo apt-get update230 sudo apt-get install -y cmake231 232 - name: Build233 id: cmake_build234 run: |235 cmake -B build -DGGML_SCHED_NO_REALLOC=ON -DGGML_CPU_KLEIDIAI=ON -DLLAMA_FATAL_WARNINGS=ON236 cmake --build build --config Release -j $(nproc) --target llama-server237 238 - name: Python setup239 id: setup_python240 run: |241 cd tools/server/tests242 python3 -m venv venv243 source venv/bin/activate244 pip install -r requirements.txt245 246 - name: Tests247 id: server_integration_tests248 if: ${{ !github.event.pull_request }}249 run: |250 cd tools/server/tests251 source venv/bin/activate252 ./tests.sh253 