Team Ai
Modelpublic

Felipe97/llama-cpp-compiled

sourceHugging Faceupdated 19d agoView on Hugging Face
0likes1.2kdownloads
server.yml214 linesDownload Raw Back to workflows
1name: Server2 3on:4  workflow_dispatch: # allows manual triggering5    inputs:6      sha:7        description: 'Commit SHA1 to build'8        required: false9        type: string10      slow_tests:11        description: 'Run slow tests'12        required: true13        type: boolean14  push:15    branches:16      - master17    paths: [18      '.github/workflows/server.yml',19      '**/CMakeLists.txt',20      '**/Makefile',21      '**/*.h',22      '**/*.hpp',23      '**/*.c',24      '**/*.cpp',25      '**/*.cu',26      '**/*.swift',27      '**/*.m',28      'tools/server/**.*'29    ]30  pull_request:31    types: [opened, synchronize, reopened]32    paths: [33      '.github/workflows/server.yml',34      '**/CMakeLists.txt',35      '**/Makefile',36      '**/*.h',37      '**/*.hpp',38      '**/*.c',39      '**/*.cpp',40      '**/*.cu',41      '**/*.swift',42      '**/*.m',43      'tools/server/**.*'44    ]45 46env:47  LLAMA_ARG_LOG_COLORS: 148  LLAMA_ARG_LOG_PREFIX: 149  LLAMA_ARG_LOG_TIMESTAMPS: 150  LLAMA_ARG_LOG_VERBOSITY: 1051 52concurrency:53  group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}54  cancel-in-progress: true55 56jobs:57  ubuntu:58    runs-on: ubuntu-24.04-arm59 60    steps:61      - name: Dependencies62        id: depends63        run: |64          sudo apt-get update65          sudo apt-get -y install \66            build-essential \67            xxd \68            git \69            cmake \70            curl \71            wget \72            language-pack-en \73            libssl-dev74 75      - name: Clone76        id: checkout77        uses: actions/checkout@v678        with:79          fetch-depth: 080          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}81 82      - name: ccache83        uses: ggml-org/ccache-action@v1.2.2484        with:85          key: server-ubuntu-24.04-arm86          save: false87 88      - name: ccache-buckets-restore89        uses: ./.github/actions/ccache-buckets90        env:91          HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}92        with:93          key: server-ubuntu-24.04-arm94          folder: llama.cpp95          hf_bucket: ggml-org/cache96 97      - name: Build98        id: cmake_build99        run: |100          cmake -B build \101            -DGGML_SCHED_NO_REALLOC=ON102          cmake --build build --config Release -j $(nproc) --target llama-server103 104      - name: ccache-buckets-save105        if: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}106        uses: ./.github/actions/ccache-buckets107        env:108          HF_TOKEN: ${{ secrets.HF_TOKEN_CACHE_OUTPUT }}109        with:110          key: server-ubuntu-24.04-arm111          folder: llama.cpp112          evict-old-files: 1d113          hf_bucket: ggml-org/cache114          save: true115 116      - name: Python setup117        id: setup_python118        uses: actions/setup-python@v6119        with:120          python-version: '3.11'121          pip-install: -r tools/server/tests/requirements.txt122 123      - name: Tests124        id: server_integration_tests125        run: |126          cd tools/server/tests127          ./tests.sh128 129      - name: Slow tests130        id: server_integration_tests_slow131        if: ${{ github.event.schedule || github.event.inputs.slow_tests == 'true' }}132        run: |133          cd tools/server/tests134          SLOW_TESTS=1 ./tests.sh135 136      - name: Tests (Backend sampling)137        id: server_integration_tests_backend_sampling138        run: |139          cd tools/server/tests140          export LLAMA_ARG_BACKEND_SAMPLING=1141          ./tests.sh142 143      - name: Slow tests (Backend sampling)144        id: server_integration_tests_slow_backend_sampling145        if: ${{ github.event.schedule || github.event.inputs.slow_tests == 'true' }}146        run: |147          cd tools/server/tests148          export LLAMA_ARG_BACKEND_SAMPLING=1149          SLOW_TESTS=1 ./tests.sh150 151  windows:152    runs-on: windows-2025153 154    steps:155      - name: Clone156        id: checkout157        uses: actions/checkout@v6158        with:159          fetch-depth: 0160          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}161 162      - name: ccache163        uses: ggml-org/ccache-action@v1.2.24164        with:165          key: server-windows-2025-x64166          evict-old-files: 1d167          save: ${{ github.event_name == 'push' && github.ref == 'refs/heads/master' }}168 169      - name: Build170        id: cmake_build171        shell: cmd172        run: |173          cmake -B build -G "Ninja Multi-Config" ^174            -DCMAKE_TOOLCHAIN_FILE=cmake/x64-windows-llvm.cmake ^175            -DCMAKE_BUILD_TYPE=Release ^176            -DLLAMA_BUILD_BORINGSSL=ON ^177            -DGGML_SCHED_NO_REALLOC=ON178          set /A NINJA_JOBS=%NUMBER_OF_PROCESSORS%-1179          cmake --build build --config Release -j %NINJA_JOBS% --target llama-server180 181      - name: Python setup182        id: setup_python183        uses: actions/setup-python@v6184        with:185          python-version: '3.11'186          pip-install: -r tools/server/tests/requirements.txt187 188      - name: Tests189        id: server_integration_tests190        shell: bash191        run: |192          cd tools/server/tests193          export PYTHONIOENCODING=":replace"194          ./tests.sh195 196      - name: Slow tests197        id: server_integration_tests_slow198        if: ${{ github.event.schedule || github.event.inputs.slow_tests == 'true' }}199        shell: bash200        run: |201          cd tools/server/tests202          export SLOW_TESTS="1"203          ./tests.sh204 205      - name: ccache-clear206        uses: ./.github/actions/ccache-clear207        env:208          GH_TOKEN: ${{ github.token }}209        with:210          key: server-windows-2025-x64211          older: 5m212          min: 1213          dry-run: ${{ github.event_name != 'push' || github.ref != 'refs/heads/master' }}214