Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1name: Server (self-hosted)2 3on:4 workflow_dispatch: # allows manual triggering5 inputs:6 sha:7 description: 'Commit SHA1 to build'8 required: false9 type: string10 slow_tests:11 description: 'Run slow tests'12 required: true13 type: boolean14 push:15 branches:16 - master17 paths: [18 '.github/workflows/server-self-hosted.yml',19 '**/CMakeLists.txt',20 '**/Makefile',21 '**/*.h',22 '**/*.hpp',23 '**/*.c',24 '**/*.cpp',25 '**/*.cu',26 '**/*.swift',27 '**/*.m',28 'tools/server/**.*'29 ]30 31env:32 # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)33 HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}34 LLAMA_ARG_LOG_COLORS: 135 LLAMA_ARG_LOG_PREFIX: 136 LLAMA_ARG_LOG_TIMESTAMPS: 137 LLAMA_ARG_LOG_VERBOSITY: 1038 39concurrency:40 group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}41 cancel-in-progress: true42 43jobs:44 server-metal:45 runs-on: [self-hosted, llama-server, macOS, ARM64]46 47 steps:48 - name: Clone49 id: checkout50 uses: actions/checkout@v651 with:52 fetch-depth: 053 ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}54 55 - name: Build56 id: cmake_build57 run: |58 cmake -B build -DGGML_SCHED_NO_REALLOC=ON59 cmake --build build --config Release -j $(sysctl -n hw.logicalcpu) --target llama-server60 61 - name: Python setup62 id: setup_python63 run: |64 cd tools/server/tests65 python3 -m venv venv66 source venv/bin/activate67 pip install -r requirements.txt68 69 - name: Tests (GPUx1)70 id: server_integration_tests71 if: ${{ !github.event.pull_request }}72 run: |73 cd tools/server/tests74 source venv/bin/activate75 pytest -v -x -m "not slow"76 77 - name: Tests (GPUx1, backend-sampling)78 id: server_integration_tests_backend_sampling79 if: ${{ !github.event.pull_request }}80 run: |81 cd tools/server/tests82 source venv/bin/activate83 export LLAMA_ARG_BACKEND_SAMPLING=184 pytest -v -x -m "not slow"85 86 - name: Tests (GPUx2)87 id: server_integration_tests_gpu288 if: ${{ !github.event.pull_request }}89 run: |90 cd tools/server/tests91 source venv/bin/activate92 export GGML_METAL_DEVICES=293 pytest -v -x -m "not slow"94 95 - name: Tests (GPUx2, backend-sampling)96 id: server_integration_tests_gpu2_backend_sampling97 if: ${{ !github.event.pull_request }}98 run: |99 cd tools/server/tests100 source venv/bin/activate101 export GGML_METAL_DEVICES=2 LLAMA_ARG_BACKEND_SAMPLING=1102 pytest -v -x -m "not slow"103 104 server-cuda:105 runs-on: [self-hosted, llama-server, Linux, NVIDIA]106 107 steps:108 - name: Clone109 id: checkout110 uses: actions/checkout@v6111 with:112 fetch-depth: 0113 ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}114 115 - name: Build116 id: cmake_build117 run: |118 cmake -B build -DGGML_CUDA=ON -DGGML_SCHED_NO_REALLOC=ON119 cmake --build build --config Release -j $(nproc) --target llama-server120 121 - name: Python setup122 id: setup_python123 run: |124 cd tools/server/tests125 python3 -m venv venv126 source venv/bin/activate127 pip install -r requirements.txt128 129 - name: Tests (GPUx1)130 id: server_integration_tests131 if: ${{ !github.event.pull_request }}132 run: |133 cd tools/server/tests134 source venv/bin/activate135 pytest -v -x -m "not slow"136 137 - name: Tests (GPUx1, backend-sampling)138 id: server_integration_tests_backend_sampling139 if: ${{ !github.event.pull_request }}140 run: |141 cd tools/server/tests142 source venv/bin/activate143 export LLAMA_ARG_BACKEND_SAMPLING=1144 pytest -v -x -m "not slow"145 146 - name: Tests (GPUx2)147 id: server_integration_tests_gpu2148 if: ${{ !github.event.pull_request }}149 run: |150 cd tools/server/tests151 source venv/bin/activate152 export GGML_CUDA_DEVICES=2153 pytest -v -x -m "not slow"154 155 - name: Tests (GPUx2, backend-sampling)156 id: server_integration_tests_gpu2_backend_sampling157 if: ${{ !github.event.pull_request }}158 run: |159 cd tools/server/tests160 source venv/bin/activate161 export GGML_CUDA_DEVICES=2 LLAMA_ARG_BACKEND_SAMPLING=1162 pytest -v -x -m "not slow"163 164 server-kleidiai:165 runs-on: ah-ubuntu_22_04-c8g_8x166 167 steps:168 - name: Clone169 id: checkout170 uses: actions/checkout@v6171 with:172 fetch-depth: 0173 ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}174 175 - name: Dependencies176 id: depends177 run: |178 set -euxo pipefail179 sudo apt-get update180 sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \181 apt-get install -y \182 build-essential \183 libssl-dev \184 python3-venv \185 gpg \186 wget \187 time \188 git-lfs189 190 git lfs install191 192 # install the latest cmake193 sudo install -d /usr/share/keyrings194 wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \195 | gpg --dearmor \196 | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null197 echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \198 | sudo tee /etc/apt/sources.list.d/kitware.list199 sudo apt-get update200 sudo apt-get install -y cmake201 202 - name: Build203 id: cmake_build204 run: |205 cmake -B build -DGGML_SCHED_NO_REALLOC=ON -DGGML_CPU_KLEIDIAI=ON206 cmake --build build --config Release -j $(nproc) --target llama-server207 208 - name: Python setup209 id: setup_python210 run: |211 cd tools/server/tests212 python3 -m venv venv213 source venv/bin/activate214 pip install -r requirements.txt215 216 - name: Tests217 id: server_integration_tests218 if: ${{ !github.event.pull_request }}219 run: |220 cd tools/server/tests221 source venv/bin/activate222 pytest -v -x -m "not slow"223 