Brunobkr/llama.cpp_AlgMor24_github
ΩFFFΣLLIa • llama.cpp • AlgMor24 ██████╗ ███████╗███████╗███████╗██╗ ██╗ ██╗ █████╗ ██╔═══██╗██╔════╝██╔════╝██╔════╝██║ ██║ ██║██╔══██╗ ██║ ██║█████╗ █████╗ █████╗ ██║ ██║ ██║███████║ ██║ ██║██╔══╝ ██╔══╝ ██╔══╝ ██║ ██║ ██║██╔══██║ ╚██████╔╝██║ ██║ ███████╗███████╗███████╗██║██║ ██║ ╚═════╝ ╚═╝ ╚═╝ ╚══════╝╚══════╝╚══════╝╚═╝╚═╝ ╚═╝ High-Performance LLM / VLM Inference & Autonomous Agentic Ecosystem… See the full description on the dataset page: https://huggingface.co/datasets/Brunobkr/llama.cpp_AlgMor24_github.
03.1k
1#!/usr/bin/env bash2#3# sample usage:4#5# mkdir tmp6#7# # CPU-only build8# bash ./ci/run.sh ./tmp/results ./tmp/mnt9#10# # with CUDA support11# GG_BUILD_CUDA=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt12#13# # with ROCm support14# GG_BUILD_ROCM=1 GG_BUILD_AMDGPU_TARGETS=gfx1151 bash ./ci/run.sh ./tmp/results ./tmp/mnt15#16# # with SYCL support17# GG_BUILD_SYCL=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt18#19# # with VULKAN support20# GG_BUILD_VULKAN=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt21#22# # with WebGPU support23# GG_BUILD_WEBGPU=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt24#25# # with MUSA support26# GG_BUILD_MUSA=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt27#28# # with KLEIDIAI support29# GG_BUILD_KLEIDIAI=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt30#31# # with BLAS support32# GG_BUILD_BLAS=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt33#34# with BLAS support (custom vendor)35# GG_BUILD_BLAS=1 GG_BUILD_BLAS_VENDOR=Intel10_64lp bash ./ci/run.sh ./tmp/results ./tmp/mnt36#37# with OPENVINO support38# GG_BUILD_OPENVINO=1 GG_BUILD_LOW_PERF=1 GGML_OPENVINO_DEVICE=CPU bash ./ci/run.sh ./tmp/results ./tmp/mnt39#40 41if [ -z "$2" ]; then42 echo "usage: $0 <output-dir> <mnt-dir>"43 exit 144fi45 46mkdir -p "$1"47mkdir -p "$2"48 49OUT=$(realpath "$1")50MNT=$(realpath "$2")51 52rm -f $OUT/*.log53rm -f $OUT/*.exit54rm -f $OUT/*.md55 56sd=`dirname $0`57cd $sd/../58SRC=`pwd`59 60CMAKE_EXTRA="-DLLAMA_FATAL_WARNINGS=${LLAMA_FATAL_WARNINGS:-ON} -DLLAMA_OPENSSL=OFF -DGGML_SCHED_NO_REALLOC=ON"61CTEST_EXTRA=""62 63# Default to use make unless specified for compatibility64CMAKE_GENERATOR="Unix Makefiles"65 66if [ ! -z "${GG_BUILD_NINJA}" ]; then67 CMAKE_GENERATOR="Ninja"68fi69 70if [ ! -z ${GG_BUILD_METAL} ]; then71 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_METAL=ON"72else73 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_METAL=OFF"74fi75 76if [ ! -z ${GG_BUILD_CUDA} ]; then77 # TODO: Remove GGML_CUDA_CUB_3DOT2 flag once CCCL 3.2 is bundled within CTK and that CTK version is used in this project78 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_CUDA=ON -DGGML_CUDA_CUB_3DOT2=ON"79 80 if command -v nvidia-smi >/dev/null 2>&1; then81 CUDA_ARCH=$(nvidia-smi --query-gpu=compute_cap --format=csv,noheader,nounits 2>/dev/null | head -1 | tr -d '.')82 if [[ -n "$CUDA_ARCH" && "$CUDA_ARCH" =~ ^[0-9]+$ ]]; then83 CMAKE_EXTRA="${CMAKE_EXTRA} -DCMAKE_CUDA_ARCHITECTURES=${CUDA_ARCH}"84 else85 echo "Warning: Using fallback CUDA architectures"86 CMAKE_EXTRA="${CMAKE_EXTRA} -DCMAKE_CUDA_ARCHITECTURES=61;70;75;80;86;89"87 fi88 else89 echo "Error: nvidia-smi not found, cannot build with CUDA"90 exit 191 fi92fi93 94if [ ! -z ${GG_BUILD_ROCM} ]; then95 CMAKE_EXTRA="${CMAKE_EXTRA} -DCMAKE_HIP_COMPILER=$(hipconfig -l)/clang -DGGML_HIP=ON"96 if [ -z ${GG_BUILD_AMDGPU_TARGETS} ]; then97 echo "Missing GG_BUILD_AMDGPU_TARGETS, please set it to your GPU architecture (e.g. gfx90a, gfx1100, etc.)"98 exit 199 fi100 101 CMAKE_EXTRA="${CMAKE_EXTRA} -DGPU_TARGETS=${GG_BUILD_AMDGPU_TARGETS}"102fi103 104if [ ! -z ${GG_BUILD_SYCL} ]; then105 if [ -z ${ONEAPI_ROOT} ]; then106 echo "Not detected ONEAPI_ROOT, please install oneAPI base toolkit and enable it by:"107 echo "source /opt/intel/oneapi/setvars.sh"108 exit 1109 fi110 # Use only main GPU111 export ONEAPI_DEVICE_SELECTOR="level_zero:0"112 # Enable sysman for correct memory reporting113 export ZES_ENABLE_SYSMAN=1114 # to circumvent precision issues on CPY operations115 export SYCL_PROGRAM_COMPILE_OPTIONS="-cl-fp32-correctly-rounded-divide-sqrt"116 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_SYCL=1 -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_SYCL_F16=ON"117fi118 119if [ ! -z ${GG_BUILD_VULKAN} ]; then120 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_VULKAN=1"121 122 if [[ "$OSTYPE" == "darwin"* ]]; then123 MACOS_RUNNER_CUSTOM_VULKAN_CMAKE_LOCATION="/usr/local/lib/cmake/vulkan"124 MACOS_RUNNER_CUSTOM_SPIRV_HEADERS_LOCATION="${MACOS_RUNNER_CUSTOM_VULKAN_CMAKE_LOCATION}/SPIRV-Headers/SPIRV-HeadersConfig.cmake"125 if [[ -f "${MACOS_RUNNER_CUSTOM_SPIRV_HEADERS_LOCATION}" || -h "${MACOS_RUNNER_CUSTOM_SPIRV_HEADERS_LOCATION}" ]]; then126 CMAKE_EXTRA="${CMAKE_EXTRA} -DSPIRV-Headers_DIR=${MACOS_RUNNER_CUSTOM_VULKAN_CMAKE_LOCATION}/SPIRV-Headers"127 fi128 fi129 130 # Build shared libs on Windows131 # to reduce binary size and avoid errors in library loading unit tests132 if uname -s | grep -qi nt; then133 CMAKE_EXTRA="${CMAKE_EXTRA} -DBUILD_SHARED_LIBS=ON"134 fi135fi136 137if [ ! -z ${GG_BUILD_WEBGPU} ]; then138 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_WEBGPU=1"139 140 if [ ! -z "${GG_BUILD_WEBGPU_DAWN_PREFIX}" ]; then141 if [ -z "${CMAKE_PREFIX_PATH}" ]; then142 export CMAKE_PREFIX_PATH="${GG_BUILD_WEBGPU_DAWN_PREFIX}"143 else144 export CMAKE_PREFIX_PATH="${GG_BUILD_WEBGPU_DAWN_PREFIX}:${CMAKE_PREFIX_PATH}"145 fi146 fi147 148 # For some systems, Dawn_DIR needs to be set explicitly, e.g., the lib64 path149 if [ ! -z "${GG_BUILD_WEBGPU_DAWN_DIR}" ]; then150 CMAKE_EXTRA="${CMAKE_EXTRA} -DDawn_DIR=${GG_BUILD_WEBGPU_DAWN_DIR}"151 fi152fi153 154if [ ! -z ${GG_BUILD_MUSA} ]; then155 # Use qy1 by default (MTT S80)156 MUSA_ARCH=${MUSA_ARCH:-21}157 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_MUSA=ON -DMUSA_ARCHITECTURES=${MUSA_ARCH}"158fi159 160if [ ! -z ${GG_BUILD_NO_SVE} ]; then161 # arm 9 and newer enables sve by default, adjust these flags depending on the cpu used162 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_NATIVE=OFF -DGGML_CPU_ARM_ARCH=armv8.5-a+fp16+i8mm"163fi164 165if [ -n "${GG_BUILD_KLEIDIAI}" ]; then166 echo ">>===== Enabling KleidiAI support"167 CMAKE_EXTRA="${CMAKE_EXTRA:+$CMAKE_EXTRA } -DGGML_CPU_KLEIDIAI=ON"168fi169 170if [ ! -z ${GG_BUILD_BLAS} ]; then171 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_BLAS=ON -DGGML_BLAS_VENDOR=${GG_BUILD_BLAS_VENDOR:-OpenBLAS}"172else173 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_BLAS=OFF"174fi175 176if [ ! -z ${GG_BUILD_OPENVINO} ]; then177 if [ -z ${OpenVINO_DIR} ]; then178 echo "OpenVINO_DIR not found, please install OpenVINO via archives and enable it by:"179 echo "source /opt/intel/openvino/setupvars.sh"180 exit 1181 fi182 CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_OPENVINO=ON"183 184 # TODO: fix and re-enable the `test-llama-archs` test below185 CTEST_EXTRA="-E test-llama-archs"186fi187 188## helpers189 190# download a file if it does not exist or if it is outdated191function gg_wget {192 local out=$1193 local url=$2194 195 local cwd=`pwd`196 197 mkdir -p $out198 cd $out199 200 # should not re-download if file is the same201 wget -nv -c -N $url202 203 cd $cwd204}205 206function gg_printf {207 printf -- "$@" >> $OUT/README.md208}209 210function gg_run {211 ci=$1212 213 set -o pipefail214 set -x215 216 gg_run_$ci | tee $OUT/$ci.log217 cur=$?218 echo "$cur" > $OUT/$ci.exit219 220 set +x221 set +o pipefail222 223 gg_sum_$ci224 225 ret=$((ret | cur))226}227 228## ci229 230# ctest_debug231 232function gg_run_ctest_debug {233 cd ${SRC}234 235 rm -rf build-ci-debug && mkdir build-ci-debug && cd build-ci-debug236 237 set -e238 239 # Check required binaries are installed240 gg_check_build_requirements241 242 (cmake -G "${CMAKE_GENERATOR}" -DCMAKE_BUILD_TYPE=Debug ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log243 (time cmake --build . --config Debug -j$(nproc)) 2>&1 | tee -a $OUT/${ci}-make.log244 245 (time ctest -C Debug --output-on-failure -L main -E "test-opt|test-backend-ops|test-llama-archs" ${CTEST_EXTRA}) 2>&1 | tee -a $OUT/${ci}-ctest.log246 247 set +e248}249 250function gg_sum_ctest_debug {251 gg_printf '### %s\n\n' "${ci}"252 253 gg_printf 'Runs ctest in debug mode\n'254 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"255 gg_printf '```\n'256 gg_printf '%s\n' "$(cat $OUT/${ci}-ctest.log)"257 gg_printf '```\n'258 gg_printf '\n'259}260 261# ctest_release262 263function gg_run_ctest_release {264 cd ${SRC}265 266 rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release267 268 set -e269 270 # Check required binaries are installed271 gg_check_build_requirements272 273 (cmake -G "${CMAKE_GENERATOR}" -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log274 (time cmake --build . --config Release -j$(nproc)) 2>&1 | tee -a $OUT/${ci}-make.log275 276 if [ -z ${GG_BUILD_LOW_PERF} ]; then277 (time ctest -C Release --output-on-failure -L 'main|python' ${CTEST_EXTRA}) 2>&1 | tee -a $OUT/${ci}-ctest.log278 else279 (time ctest -C Release --output-on-failure -L main -E test-opt ${CTEST_EXTRA}) 2>&1 | tee -a $OUT/${ci}-ctest.log280 fi281 282 set +e283}284 285function gg_sum_ctest_release {286 gg_printf '### %s\n\n' "${ci}"287 288 gg_printf 'Runs ctest in release mode\n'289 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"290 gg_printf '```\n'291 gg_printf '%s\n' "$(cat $OUT/${ci}-ctest.log)"292 gg_printf '```\n'293}294 295# test_scripts296 297function gg_run_test_scripts {298 cd ${SRC}299 300 set -e301 302 (cd ./tools/gguf-split && time bash tests.sh "$SRC/build-ci-release/bin" "$MNT/models") 2>&1 | tee -a $OUT/${ci}-scripts.log303 (cd ./tools/quantize && time bash tests.sh "$SRC/build-ci-release/bin" "$MNT/models") 2>&1 | tee -a $OUT/${ci}-scripts.log304 305 set +e306}307 308function gg_sum_test_scripts {309 gg_printf '### %s\n\n' "${ci}"310 311 gg_printf 'Runs test scripts\n'312 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"313 gg_printf '```\n'314 gg_printf '%s\n' "$(cat $OUT/${ci}-scripts.log)"315 gg_printf '```\n'316 gg_printf '\n'317}318 319function gg_get_model {320 #local gguf_0="$MNT/models/qwen3/0.6B/ggml-model-f16.gguf"321 local gguf_0="$MNT/models/qwen3/0.6B/ggml-model-q4_0.gguf"322 if [[ -s $gguf_0 ]]; then323 echo -n "$gguf_0"324 else325 echo >&2 "No model found. Can't run gg_run_ctest_with_model."326 exit 1327 fi328}329 330function gg_run_ctest_with_model_debug {331 cd ${SRC}332 333 local model; model=$(gg_get_model)334 cd build-ci-debug335 set -e336 337 (LLAMACPP_TEST_MODELFILE="$model" time ctest -C Debug --output-on-failure -L model) 2>&1 | tee -a $OUT/${ci}-ctest.log338 339 set +e340 cd ..341}342 343function gg_run_ctest_with_model_release {344 cd ${SRC}345 346 local model; model=$(gg_get_model)347 cd build-ci-release348 set -e349 350 (LLAMACPP_TEST_MODELFILE="$model" time ctest -C Release --output-on-failure -L model) 2>&1 | tee -a $OUT/${ci}-ctest.log351 352 # test memory leaks353 #if [[ ! -z ${GG_BUILD_METAL} ]]; then354 # # TODO: this hangs for some reason ...355 # (time leaks -quiet -atExit -- ./bin/test-thread-safety -m $model --parallel 2 -t 2 -p "hello") 2>&1 | tee -a $OUT/${ci}-leaks.log356 #fi357 358 set +e359 cd ..360}361 362function gg_sum_ctest_with_model_debug {363 gg_printf '### %s\n\n' "${ci}"364 365 gg_printf 'Runs ctest with model files in debug mode\n'366 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"367 gg_printf '```\n'368 gg_printf '%s\n' "$(cat $OUT/${ci}-ctest.log)"369 gg_printf '```\n'370}371 372function gg_sum_ctest_with_model_release {373 gg_printf '### %s\n\n' "${ci}"374 375 gg_printf 'Runs ctest with model files in release mode\n'376 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"377 gg_printf '```\n'378 gg_printf '%s\n' "$(cat $OUT/${ci}-ctest.log)"379 gg_printf '```\n'380}381 382# qwen3_0_6b383 384function gg_run_qwen3_0_6b {385 cd ${SRC}386 387 gg_wget models-mnt/qwen3/0.6B/ https://huggingface.co/Qwen/Qwen3-0.6B-Base/raw/main/config.json388 gg_wget models-mnt/qwen3/0.6B/ https://huggingface.co/Qwen/Qwen3-0.6B-Base/raw/main/tokenizer.json389 gg_wget models-mnt/qwen3/0.6B/ https://huggingface.co/Qwen/Qwen3-0.6B-Base/raw/main/tokenizer_config.json390 #gg_wget models-mnt/qwen3/0.6B/ https://huggingface.co/Qwen/Qwen3-0.6B-Base/raw/main/special_tokens_map.json391 gg_wget models-mnt/qwen3/0.6B/ https://huggingface.co/Qwen/Qwen3-0.6B-Base/resolve/main/model.safetensors392 393 394 gg_wget models-mnt/wikitext/ https://huggingface.co/datasets/ggml-org/ci/resolve/main/wikitext-2-raw-v1.zip395 unzip -o models-mnt/wikitext/wikitext-2-raw-v1.zip -d models-mnt/wikitext/396 397 path_models="../models-mnt/qwen3/0.6B"398 path_wiki="../models-mnt/wikitext/wikitext-2-raw"399 400 rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release401 402 set -e403 404 (cmake -G "${CMAKE_GENERATOR}" -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log405 (time cmake --build . --config Release -j$(nproc)) 2>&1 | tee -a $OUT/${ci}-make.log406 407 python3 ../convert_hf_to_gguf.py ${path_models} --outfile ${path_models}/ggml-model-f16.gguf --outtype f16408 python3 ../convert_hf_to_gguf.py ${path_models} --outfile ${path_models}/ggml-model-bf16.gguf --outtype bf16409 410 model_f16="${path_models}/ggml-model-f16.gguf"411 model_bf16="${path_models}/ggml-model-bf16.gguf"412 model_q8_0="${path_models}/ggml-model-q8_0.gguf"413 model_q4_0="${path_models}/ggml-model-q4_0.gguf"414 model_q4_1="${path_models}/ggml-model-q4_1.gguf"415 model_q5_0="${path_models}/ggml-model-q5_0.gguf"416 model_q5_1="${path_models}/ggml-model-q5_1.gguf"417 model_q2_k="${path_models}/ggml-model-q2_k.gguf"418 model_q3_k="${path_models}/ggml-model-q3_k.gguf"419 model_q4_k="${path_models}/ggml-model-q4_k.gguf"420 model_q5_k="${path_models}/ggml-model-q5_k.gguf"421 model_q6_k="${path_models}/ggml-model-q6_k.gguf"422 423 wiki_test="${path_wiki}/wiki.test.raw"424 425 ./bin/llama-quantize ${model_bf16} ${model_q8_0} q8_0 $(nproc)426 ./bin/llama-quantize ${model_bf16} ${model_q4_0} q4_0 $(nproc)427 ./bin/llama-quantize ${model_bf16} ${model_q4_1} q4_1 $(nproc)428 ./bin/llama-quantize ${model_bf16} ${model_q5_0} q5_0 $(nproc)429 ./bin/llama-quantize ${model_bf16} ${model_q5_1} q5_1 $(nproc)430 ./bin/llama-quantize ${model_bf16} ${model_q2_k} q2_k $(nproc)431 ./bin/llama-quantize ${model_bf16} ${model_q3_k} q3_k $(nproc)432 ./bin/llama-quantize ${model_bf16} ${model_q4_k} q4_k $(nproc)433 ./bin/llama-quantize ${model_bf16} ${model_q5_k} q5_k $(nproc)434 ./bin/llama-quantize ${model_bf16} ${model_q6_k} q6_k $(nproc)435 436 (time ./bin/llama-fit-params --model ${model_f16} 2>&1 | tee -a $OUT/${ci}-fp-f16.log)437 438 (time ./bin/llama-completion -no-cnv --model ${model_f16} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log439 (time ./bin/llama-completion -no-cnv --model ${model_bf16} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-bf16.log440 (time ./bin/llama-completion -no-cnv --model ${model_q8_0} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log441 (time ./bin/llama-completion -no-cnv --model ${model_q4_0} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_0.log442 (time ./bin/llama-completion -no-cnv --model ${model_q4_1} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_1.log443 (time ./bin/llama-completion -no-cnv --model ${model_q5_0} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_0.log444 (time ./bin/llama-completion -no-cnv --model ${model_q5_1} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_1.log445 (time ./bin/llama-completion -no-cnv --model ${model_q2_k} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q2_k.log446 (time ./bin/llama-completion -no-cnv --model ${model_q3_k} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q3_k.log447 (time ./bin/llama-completion -no-cnv --model ${model_q4_k} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_k.log448 (time ./bin/llama-completion -no-cnv --model ${model_q5_k} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_k.log449 (time ./bin/llama-completion -no-cnv --model ${model_q6_k} -ngl 99 -c 1024 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q6_k.log450 451 (time ./bin/llama-perplexity --model ${model_f16} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log452 if [ -z ${GG_BUILD_NO_BF16} ]; then453 (time ./bin/llama-perplexity --model ${model_bf16} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-bf16.log454 fi455 (time ./bin/llama-perplexity --model ${model_q8_0} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log456 (time ./bin/llama-perplexity --model ${model_q4_0} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_0.log457 (time ./bin/llama-perplexity --model ${model_q4_1} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_1.log458 (time ./bin/llama-perplexity --model ${model_q5_0} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_0.log459 (time ./bin/llama-perplexity --model ${model_q5_1} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_1.log460 (time ./bin/llama-perplexity --model ${model_q2_k} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q2_k.log461 (time ./bin/llama-perplexity --model ${model_q3_k} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q3_k.log462 (time ./bin/llama-perplexity --model ${model_q4_k} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_k.log463 (time ./bin/llama-perplexity --model ${model_q5_k} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_k.log464 (time ./bin/llama-perplexity --model ${model_q6_k} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-tg-q6_k.log465 466 (time ./bin/llama-imatrix --model ${model_f16} -f ${wiki_test} -ngl 99 -c 1024 -b 512 --chunks 2 ) 2>&1 | tee -a $OUT/${ci}-imatrix.log467 468 (time ./bin/test-save-load-state --model ${model_q4_0} -ngl 10 -c 1024 -fa off --no-op-offload) 2>&1 | tee -a $OUT/${ci}-save-load-state.log469 (time ./bin/test-save-load-state --model ${model_q4_0} -ngl 10 -c 1024 -fa on --no-op-offload) 2>&1 | tee -a $OUT/${ci}-save-load-state.log470 (time ./bin/test-save-load-state --model ${model_q4_0} -ngl 99 -c 1024 -fa off ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log471 (time ./bin/test-save-load-state --model ${model_q4_0} -ngl 99 -c 1024 -fa on ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log472 473 function check_ppl {474 qnt="$1"475 ppl=$(echo "$2" | grep -oE "[0-9]+\.[0-9]+" | tail -n 1)476 477 if [ $(echo "$ppl > 20.0" | bc) -eq 1 ]; then478 printf ' - %s @ %s (FAIL: ppl > 20.0)\n' "$qnt" "$ppl"479 return 20480 fi481 482 printf ' - %s @ %s OK\n' "$qnt" "$ppl"483 return 0484 }485 486 check_ppl "f16" "$(cat $OUT/${ci}-tg-f16.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log487 if [ -z ${GG_BUILD_NO_BF16} ]; then488 check_ppl "bf16" "$(cat $OUT/${ci}-tg-bf16.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log489 fi490 check_ppl "q8_0" "$(cat $OUT/${ci}-tg-q8_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log491 check_ppl "q4_0" "$(cat $OUT/${ci}-tg-q4_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log492 check_ppl "q4_1" "$(cat $OUT/${ci}-tg-q4_1.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log493 check_ppl "q5_0" "$(cat $OUT/${ci}-tg-q5_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log494 check_ppl "q5_1" "$(cat $OUT/${ci}-tg-q5_1.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log495 #check_ppl "q2_k" "$(cat $OUT/${ci}-tg-q2_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log # note: ppl > 20.0 for this quant and model496 check_ppl "q3_k" "$(cat $OUT/${ci}-tg-q3_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log497 check_ppl "q4_k" "$(cat $OUT/${ci}-tg-q4_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log498 check_ppl "q5_k" "$(cat $OUT/${ci}-tg-q5_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log499 check_ppl "q6_k" "$(cat $OUT/${ci}-tg-q6_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log500 501 cat $OUT/${ci}-imatrix.log | grep "Final" >> $OUT/${ci}-imatrix-sum.log502 503 set +e504}505 506function gg_sum_qwen3_0_6b {507 gg_printf '### %s\n\n' "${ci}"508 509 gg_printf 'Qwen3 0.6B:\n'510 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"511 gg_printf '- perplexity:\n%s\n' "$(cat $OUT/${ci}-ppl.log)"512 gg_printf '- imatrix:\n```\n%s\n```\n' "$(cat $OUT/${ci}-imatrix-sum.log)"513 gg_printf '- f16:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-f16.log)"514 if [ -z ${GG_BUILD_NO_BF16} ]; then515 gg_printf '- bf16:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-bf16.log)"516 fi517 gg_printf '- q8_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q8_0.log)"518 gg_printf '- q4_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_0.log)"519 gg_printf '- q4_1:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_1.log)"520 gg_printf '- q5_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_0.log)"521 gg_printf '- q5_1:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_1.log)"522 gg_printf '- q2_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q2_k.log)"523 gg_printf '- q3_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q3_k.log)"524 gg_printf '- q4_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_k.log)"525 gg_printf '- q5_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_k.log)"526 gg_printf '- q6_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q6_k.log)"527 gg_printf '- save-load-state: \n```\n%s\n```\n' "$(cat $OUT/${ci}-save-load-state.log)"528}529 530# bge-small531 532function gg_run_embd_bge_small {533 cd ${SRC}534 535 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/config.json536 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/tokenizer.json537 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/tokenizer_config.json538 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/special_tokens_map.json539 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/resolve/main/pytorch_model.bin540 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/sentence_bert_config.json541 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/vocab.txt542 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/modules.json543 gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/config.json544 545 gg_wget models-mnt/bge-small/1_Pooling https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/1_Pooling/config.json546 547 path_models="../models-mnt/bge-small"548 549 rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release550 551 set -e552 553 (cmake -G "${CMAKE_GENERATOR}" -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log554 (time cmake --build . --config Release -j$(nproc)) 2>&1 | tee -a $OUT/${ci}-make.log555 556 python3 ../convert_hf_to_gguf.py ${path_models} --outfile ${path_models}/ggml-model-f16.gguf557 558 model_f16="${path_models}/ggml-model-f16.gguf"559 model_q8_0="${path_models}/ggml-model-q8_0.gguf"560 561 ./bin/llama-quantize ${model_f16} ${model_q8_0} q8_0562 563 (time ./bin/llama-fit-params --model ${model_f16} 2>&1 | tee -a $OUT/${ci}-fp-f16.log)564 565 (time ./bin/llama-embedding --model ${model_f16} -p "I believe the meaning of life is" -ngl 99 -c 0 --no-op-offload) 2>&1 | tee -a $OUT/${ci}-tg-f16.log566 (time ./bin/llama-embedding --model ${model_q8_0} -p "I believe the meaning of life is" -ngl 99 -c 0 --no-op-offload) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log567 568 set +e569}570 571function gg_sum_embd_bge_small {572 gg_printf '### %s\n\n' "${ci}"573 574 gg_printf 'BGE Small (BERT):\n'575 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"576 gg_printf '- f16: \n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-f16.log)"577 gg_printf '- q8_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q8_0.log)"578}579 580# rerank_tiny581 582function gg_run_rerank_tiny {583 cd ${SRC}584 585 gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/config.json586 gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/tokenizer.json587 gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/tokenizer_config.json588 gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/special_tokens_map.json589 gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/resolve/main/pytorch_model.bin590 gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/vocab.json591 592 path_models="../models-mnt/rerank-tiny"593 594 rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release595 596 set -e597 598 (cmake -G "${CMAKE_GENERATOR}" -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log599 (time cmake --build . --config Release -j$(nproc)) 2>&1 | tee -a $OUT/${ci}-make.log600 601 python3 ../convert_hf_to_gguf.py ${path_models} --outfile ${path_models}/ggml-model-f16.gguf602 603 model_f16="${path_models}/ggml-model-f16.gguf"604 605 (time ./bin/llama-fit-params --model ${model_f16} 2>&1 | tee -a $OUT/${ci}-fp-f16.log)606 607 # for this model, the SEP token is "</s>"608 (time ./bin/llama-embedding --model ${model_f16} -p "what is panda?\thi\nwhat is panda?\tit's a bear\nwhat is panda?\tThe giant panda (Ailuropoda melanoleuca), sometimes called a panda bear or simply panda, is a bear species endemic to China." -ngl 99 -c 0 --pooling rank --embd-normalize -1 --no-op-offload --verbose-prompt) 2>&1 | tee -a $OUT/${ci}-rk-f16.log609 610 # sample output611 # rerank score 0: 0.029612 # rerank score 1: 0.029613 # rerank score 2: 0.135614 615 # check that the score is in the range [$3, $4]616 function check_score {617 qnt="$1"618 score=$(echo "$2" | grep -oE "[0-9]+\.[0-9]+" | tail -n 1)619 620 if [ $(echo "$score < $3" | bc) -eq 1 ] || [ $(echo "$score > $4" | bc) -eq 1 ]; then621 printf ' - %s @ %s (FAIL: score not in range [%s, %s])\n' "$qnt" "$score" "$3" "$4"622 return 20623 fi624 625 printf ' - %s @ %s OK\n' "$qnt" "$score"626 return 0627 }628 629 check_score "rerank score 0" "$(cat $OUT/${ci}-rk-f16.log | grep "rerank score 0")" "0.00" "0.05" | tee -a $OUT/${ci}-rk-f16.log630 check_score "rerank score 1" "$(cat $OUT/${ci}-rk-f16.log | grep "rerank score 1")" "0.00" "0.05" | tee -a $OUT/${ci}-rk-f16.log631 check_score "rerank score 2" "$(cat $OUT/${ci}-rk-f16.log | grep "rerank score 2")" "0.10" "0.30" | tee -a $OUT/${ci}-rk-f16.log632 633 set +e634}635 636function gg_sum_rerank_tiny {637 gg_printf '### %s\n\n' "${ci}"638 639 gg_printf 'Rerank Tiny (Jina):\n'640 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"641 gg_printf '- f16: \n```\n%s\n```\n' "$(cat $OUT/${ci}-rk-f16.log)"642}643 644function gg_check_build_requirements {645 if ! command -v git &> /dev/null; then646 gg_printf 'git not found, please install\n'647 exit 1648 fi649 650 if ! command -v git-lfs &> /dev/null; then651 gg_printf 'git-lfs not found, please install\n'652 exit 1653 fi654 655 if ! git config --get filter.lfs.clean &> /dev/null; then656 gg_printf 'git-lfs not initialized, please run `git lfs install`\n'657 exit 1658 fi659 660 if ! command -v wget &> /dev/null; then661 gg_printf 'wget not found, please install\n'662 exit 1663 fi664 665 if ! command -v python3 &> /dev/null; then666 gg_printf 'python3 not found, please install\n'667 exit 1668 fi669 670 if ! command -v pip3 &> /dev/null; then671 gg_printf 'pip3 not found, please install\n'672 exit 1673 fi674 675 if ! python3 -m ensurepip --help &> /dev/null; then676 gg_printf 'ensurepip not found, please install python3-venv package\n'677 exit 1678 fi679 680 if ! command -v cmake &> /dev/null; then681 gg_printf 'cmake not found, please install\n'682 exit 1683 fi684 685 if ! command -v ccache &> /dev/null; then686 gg_printf 'ccache not found, please consider installing for faster builds\n'687 fi688 689 if ! command -v ctest &> /dev/null; then690 gg_printf 'ctest not found, please install\n'691 exit 1692 fi693}694 695function gg_run_test_backend_ops_cpu {696 cd ${SRC}697 698 cd build-ci-release699 700 set -e701 702 (time ./bin/test-backend-ops -b CPU ) 2>&1 | tee -a $OUT/${ci}-test-backend-ops-cpu.log703 704 set +e705}706 707function gg_sum_test_backend_ops_cpu {708 gg_printf '### %s\n\n' "${ci}"709 710 gg_printf 'Runs test-backend-ops for CPU backend\n'711 gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"712 gg_printf '```\n'713 gg_printf '%s\n' "$(cat $OUT/${ci}-test-backend-ops-cpu.log)"714 gg_printf '```\n'715 gg_printf '\n'716}717 718## main719 720export LLAMA_ARG_LOG_PREFIX=1721export LLAMA_ARG_LOG_TIMESTAMPS=1722 723if [ -z ${GG_BUILD_LOW_PERF} ]; then724 # Create symlink: ./llama.cpp/models-mnt -> $MNT/models725 rm -rf ${SRC}/models-mnt726 mnt_models=${MNT}/models727 mkdir -p ${mnt_models}728 ln -sfn ${mnt_models} ${SRC}/models-mnt729 730 # Create a fresh python3 venv and enter it731 if ! python3 -m venv "$MNT/venv"; then732 echo "Error: Failed to create Python virtual environment at $MNT/venv."733 exit 1734 fi735 source "$MNT/venv/bin/activate"736 737 pip install -r ${SRC}/requirements.txt --disable-pip-version-check738 pip install --editable gguf-py --disable-pip-version-check739fi740 741ret=0742 743test $ret -eq 0 && gg_run ctest_debug744test $ret -eq 0 && gg_run ctest_release745 746if [ ! -z ${GG_BUILD_HIGH_PERF} ]; then747 test $ret -eq 0 && gg_run test_backend_ops_cpu748fi749 750if [ -z ${GG_BUILD_LOW_PERF} ]; then751 test $ret -eq 0 && gg_run embd_bge_small752 test $ret -eq 0 && gg_run rerank_tiny753 754 if [ -z ${GG_BUILD_CLOUD} ] || [ ${GG_BUILD_EXTRA_TESTS_0} ]; then755 test $ret -eq 0 && gg_run test_scripts756 fi757 758 test $ret -eq 0 && gg_run qwen3_0_6b759 760 test $ret -eq 0 && gg_run ctest_with_model_debug761 test $ret -eq 0 && gg_run ctest_with_model_release762fi763 764cat $OUT/README.md765 766exit $ret767 