Team Ai
Datasetpublic

Brunobkr/llama.cpp_AlgMor24_github

ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.

sourceHugging Faceupdated 2mo agoView on Hugging Face
0likes3.1kdownloads
server-self-hosted.yml223 linesDownload Raw Back to workflows
1name: Server (self-hosted)2 3on:4  workflow_dispatch: # allows manual triggering5    inputs:6      sha:7        description: 'Commit SHA1 to build'8        required: false9        type: string10      slow_tests:11        description: 'Run slow tests'12        required: true13        type: boolean14  push:15    branches:16      - master17    paths: [18      '.github/workflows/server-self-hosted.yml',19      '**/CMakeLists.txt',20      '**/Makefile',21      '**/*.h',22      '**/*.hpp',23      '**/*.c',24      '**/*.cpp',25      '**/*.cu',26      '**/*.swift',27      '**/*.m',28      'tools/server/**.*'29    ]30 31env:32  # note: this is dud token to avoid rate limiting (https://github.com/ggml-org/llama.cpp/pull/25706#issuecomment-4979941302)33  HF_TOKEN: ${{ secrets.HF_TOKEN_CI }}34  LLAMA_ARG_LOG_COLORS: 135  LLAMA_ARG_LOG_PREFIX: 136  LLAMA_ARG_LOG_TIMESTAMPS: 137  LLAMA_ARG_LOG_VERBOSITY: 1038 39concurrency:40  group: ${{ github.workflow }}-${{ github.ref }}-${{ github.head_ref || github.run_id }}41  cancel-in-progress: true42 43jobs:44  server-metal:45    runs-on: [self-hosted, llama-server, macOS, ARM64]46 47    steps:48      - name: Clone49        id: checkout50        uses: actions/checkout@v651        with:52          fetch-depth: 053          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}54 55      - name: Build56        id: cmake_build57        run: |58          cmake -B build -DGGML_SCHED_NO_REALLOC=ON59          cmake --build build --config Release -j $(sysctl -n hw.logicalcpu) --target llama-server60 61      - name: Python setup62        id: setup_python63        run: |64          cd tools/server/tests65          python3 -m venv venv66          source venv/bin/activate67          pip install -r requirements.txt68 69      - name: Tests (GPUx1)70        id: server_integration_tests71        if: ${{ !github.event.pull_request }}72        run: |73          cd tools/server/tests74          source venv/bin/activate75          pytest -v -x -m "not slow"76 77      - name: Tests (GPUx1, backend-sampling)78        id: server_integration_tests_backend_sampling79        if: ${{ !github.event.pull_request }}80        run: |81          cd tools/server/tests82          source venv/bin/activate83          export LLAMA_ARG_BACKEND_SAMPLING=184          pytest -v -x -m "not slow"85 86      - name: Tests (GPUx2)87        id: server_integration_tests_gpu288        if: ${{ !github.event.pull_request }}89        run: |90          cd tools/server/tests91          source venv/bin/activate92          export GGML_METAL_DEVICES=293          pytest -v -x -m "not slow"94 95      - name: Tests (GPUx2, backend-sampling)96        id: server_integration_tests_gpu2_backend_sampling97        if: ${{ !github.event.pull_request }}98        run: |99          cd tools/server/tests100          source venv/bin/activate101          export GGML_METAL_DEVICES=2 LLAMA_ARG_BACKEND_SAMPLING=1102          pytest -v -x -m "not slow"103 104  server-cuda:105    runs-on: [self-hosted, llama-server, Linux, NVIDIA]106 107    steps:108      - name: Clone109        id: checkout110        uses: actions/checkout@v6111        with:112          fetch-depth: 0113          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}114 115      - name: Build116        id: cmake_build117        run: |118          cmake -B build -DGGML_CUDA=ON -DGGML_SCHED_NO_REALLOC=ON119          cmake --build build --config Release -j $(nproc) --target llama-server120 121      - name: Python setup122        id: setup_python123        run: |124          cd tools/server/tests125          python3 -m venv venv126          source venv/bin/activate127          pip install -r requirements.txt128 129      - name: Tests (GPUx1)130        id: server_integration_tests131        if: ${{ !github.event.pull_request }}132        run: |133          cd tools/server/tests134          source venv/bin/activate135          pytest -v -x -m "not slow"136 137      - name: Tests (GPUx1, backend-sampling)138        id: server_integration_tests_backend_sampling139        if: ${{ !github.event.pull_request }}140        run: |141          cd tools/server/tests142          source venv/bin/activate143          export LLAMA_ARG_BACKEND_SAMPLING=1144          pytest -v -x -m "not slow"145 146      - name: Tests (GPUx2)147        id: server_integration_tests_gpu2148        if: ${{ !github.event.pull_request }}149        run: |150          cd tools/server/tests151          source venv/bin/activate152          export GGML_CUDA_DEVICES=2153          pytest -v -x -m "not slow"154 155      - name: Tests (GPUx2, backend-sampling)156        id: server_integration_tests_gpu2_backend_sampling157        if: ${{ !github.event.pull_request }}158        run: |159          cd tools/server/tests160          source venv/bin/activate161          export GGML_CUDA_DEVICES=2 LLAMA_ARG_BACKEND_SAMPLING=1162          pytest -v -x -m "not slow"163 164  server-kleidiai:165    runs-on: ah-ubuntu_22_04-c8g_8x166 167    steps:168      - name: Clone169        id: checkout170        uses: actions/checkout@v6171        with:172          fetch-depth: 0173          ref: ${{ github.event.inputs.sha || github.event.pull_request.head.sha || github.sha || github.head_ref || github.ref_name }}174 175      - name: Dependencies176        id: depends177        run: |178          set -euxo pipefail179          sudo apt-get update180          sudo DEBIAN_FRONTEND=noninteractive NEEDRESTART_MODE=a \181          apt-get install -y \182           build-essential \183           libssl-dev \184           python3-venv \185           gpg \186           wget \187           time \188           git-lfs189 190          git lfs install191 192          # install the latest cmake193          sudo install -d /usr/share/keyrings194          wget -O - https://apt.kitware.com/keys/kitware-archive-latest.asc \195           | gpg --dearmor \196           | sudo tee /usr/share/keyrings/kitware-archive-keyring.gpg >/dev/null197          echo 'deb [signed-by=/usr/share/keyrings/kitware-archive-keyring.gpg] https://apt.kitware.com/ubuntu/ jammy main' \198           | sudo tee /etc/apt/sources.list.d/kitware.list199          sudo apt-get update200          sudo apt-get install -y cmake201 202      - name: Build203        id: cmake_build204        run: |205          cmake -B build -DGGML_SCHED_NO_REALLOC=ON -DGGML_CPU_KLEIDIAI=ON206          cmake --build build --config Release -j $(nproc) --target llama-server207 208      - name: Python setup209        id: setup_python210        run: |211          cd tools/server/tests212          python3 -m venv venv213          source venv/bin/activate214          pip install -r requirements.txt215 216      - name: Tests217        id: server_integration_tests218        if: ${{ !github.event.pull_request }}219        run: |220          cd tools/server/tests221          source venv/bin/activate222          pytest -v -x -m "not slow"223 
Brunobkr/llama.cpp_AlgMor24_github · Team Ai