Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 21d agoView on Hugging Face
0likes1.2kdownloads
server-self-hosted.yml253 linesDownload Raw Back to workflows
1name: Server (self-hosted)2 3on:4  workflow_dispatch: # allows manual triggering5    inputs:6      sha:7        description: 'Commit SHA1 to build'8        required: false9        type: string10      slow_tests:11        description: 'Run slow tests'12        required: true13        type: boolean14  push:15    branches:16      - master17    paths: [18      '.github/workflows/server-self-hosted.yml',19      '**/CMakeLists.txt',20      '**/Makefile',21      '**/*.h',22      '**/*.hpp',23      '**/*.c',24      '**/*.cpp',25      '**/*.cu',26      '**/*.swift',27      '**/*.m',28      'tools/server/**.*'29    ]30 31env:32  # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)33  HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}34  LLAMA_ARG_LOG_COLORS: 135  LLAMA_ARG_LOG_PREFIX: 136  LLAMA_ARG_LOG_TIMESTAMPS: 137  LLAMA_ARG_LOG_VERBOSITY: 1038 39concurrency:40  group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}41  cancel-in-progress: true42 43jobs:44  server-metal:45    runs-on: [self-hosted, llama-server, macOS, ARM64]46 47    steps:48      - name: Clone49        id: checkout50        uses: actions/checkout@v651        with:52          fetch-depth: 053          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}54 55      - name: Build56        id: cmake_build57        run: |58          cmake -B build -DGGML_SCHED_NO_REALLOC=ON59          cmake --build build --config Release -j $(sysctl -n hw.logicalcpu) --target llama-server60 61      - name: Python setup62        id: setup_python63        run: |64          cd tools/server/tests65          python3 -m venv venv66          source venv/bin/activate67          pip install -r requirements.txt68 69      - name: Tests (GPUx1)70        id: server_integration_tests71        if: ${{ !github.event.pull_request }}72        run: |73          cd tools/server/tests74          source venv/bin/activate75          PYTEST_WORKERS=1 ./tests.sh76 77      - name: Tests (GPUx1, backend-sampling)78        id: server_integration_tests_backend_sampling79        if: ${{ !github.event.pull_request }}80        run: |81          cd tools/server/tests82          source venv/bin/activate83          export LLAMA_ARG_BACKEND_SAMPLING=184          PYTEST_WORKERS=1 ./tests.sh85 86      - name: Tests (GPUx2)87        id: server_integration_tests_gpu288        if: ${{ !github.event.pull_request }}89        run: |90          cd tools/server/tests91          source venv/bin/activate92          export GGML_METAL_DEVICES=293          PYTEST_WORKERS=1 ./tests.sh94 95      - name: Tests (GPUx2, backend-sampling)96        id: server_integration_tests_gpu2_backend_sampling97        if: ${{ !github.event.pull_request }}98        run: |99          cd tools/server/tests100          source venv/bin/activate101          export GGML_METAL_DEVICES=2 LLAMA_ARG_BACKEND_SAMPLING=1102          PYTEST_WORKERS=1 ./tests.sh103 104  server-cuda:105    runs-on: "hf-jobs-t4-small:cuda13"106 107    steps:108      - name: Clone109        id: checkout110        uses: actions/checkout@v6111        with:112          fetch-depth: 0113          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}114 115      - name: Install dependencies116        run: |117          sudo apt update118          sudo apt install -y cmake libssl-dev python3 python3-venv python3-pip119 120      - name: ccache121        uses: ggml-org/ccache-action@v1.2.24122        with:123          restore: false124          save: false125 126      - name: ccache-buckets-restore127        uses: ./.github/actions/ccache-buckets128        with:129          key: self-hosted-server-cuda130          folder: llama.cpp131          hf_bucket: ggml-org/cache132 133      - name: Build134        id: cmake_build135        run: |136          cmake -B build -DGGML_CUDA=ON -DGGML_SCHED_NO_REALLOC=ON -DCMAKE_CUDA_COMPILER=/usr/local/cuda/bin/nvcc137          cmake --build build --config Release -j $(nproc) --target llama-server138 139      - name: ccache-buckets-save140        if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}141        uses: ./.github/actions/ccache-buckets142        env:143          HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}144        with:145          key: self-hosted-server-cuda146          folder: llama.cpp147          evict-old-files: 1d148          hf_bucket: ggml-org/cache149          save: true150 151      - name: Python setup152        id: setup_python153        run: |154          cd tools/server/tests155          python3 -m venv venv156          source venv/bin/activate157          pip install -r requirements.txt158 159      - name: Tests (GPUx1)160        id: server_integration_tests161        if: ${{ !github.event.pull_request }}162        run: |163          cd tools/server/tests164          source venv/bin/activate165          PYTEST_WORKERS=1 ./tests.sh166 167      - name: Tests (GPUx1, backend-sampling)168        id: server_integration_tests_backend_sampling169        if: ${{ !github.event.pull_request }}170        run: |171          cd tools/server/tests172          source venv/bin/activate173          export LLAMA_ARG_BACKEND_SAMPLING=1174          PYTEST_WORKERS=1 ./tests.sh175 176      - name: Tests (GPUx2)177        id: server_integration_tests_gpu2178        if: ${{ !github.event.pull_request }}179        run: |180          cd tools/server/tests181          source venv/bin/activate182          export GGML_CUDA_DEVICES=2183          PYTEST_WORKERS=1 ./tests.sh184 185      - name: Tests (GPUx2, backend-sampling)186        id: server_integration_tests_gpu2_backend_sampling187        if: ${{ !github.event.pull_request }}188        run: |189          cd tools/server/tests190          source venv/bin/activate191          export GGML_CUDA_DEVICES=2 LLAMA_ARG_BACKEND_SAMPLING=1192          PYTEST_WORKERS=1 ./tests.sh193 194  server-kleidiai:195    runs-on: ah-ubuntu_24_04-c8g_8x196 197    steps:198      - name: Clone199        id: checkout200        uses: actions/checkout@v6201        with:202          fetch-depth: 0203          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}204 205      - name: Dependencies206        id: depends207        run: |208          set -euxo pipefail209          sudo apt-get update210          sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \211          apt-get install -y \212           build-essential \213           libssl-dev \214           python3-venv \215           gpg \216           wget \217           time \218           git-lfs219 220          git lfs install221 222          # install the latest cmake223          sudo install -d /usr/share/keyrings224          wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \225           | gpg --dearmor \226           | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null227          echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \228           | sudo tee /etc/apt/sources.list.d/kitware.list229          sudo apt-get update230          sudo apt-get install -y cmake231 232      - name: Build233        id: cmake_build234        run: |235          cmake -B build -DGGML_SCHED_NO_REALLOC=ON -DGGML_CPU_KLEIDIAI=ON -DLLAMA_FATAL_WARNINGS=ON236          cmake --build build --config Release -j $(nproc) --target llama-server237 238      - name: Python setup239        id: setup_python240        run: |241          cd tools/server/tests242          python3 -m venv venv243          source venv/bin/activate244          pip install -r requirements.txt245 246      - name: Tests247        id: server_integration_tests248        if: ${{ !github.event.pull_request }}249        run: |250          cd tools/server/tests251          source venv/bin/activate252          ./tests.sh253