Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
run.sh855 linesDownload Raw Back to ci
1#!/bin/bash2#3# sample usage:4#5# mkdir tmp6#7# # CPU-only build8# bash ./ci/run.sh ./tmp/results ./tmp/mnt9#10# # with CUDA support11# GG_BUILD_CUDA=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt12#13# # with SYCL support14# GG_BUILD_SYCL=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt15#16# # with VULKAN support17# GG_BUILD_VULKAN=1 bash ./ci/run.sh ./tmp/results ./tmp/mnt18#19 20if [ -z "$2" ]; then21    echo "usage: $0 <output-dir> <mnt-dir>"22    exit 123fi24 25mkdir -p "$1"26mkdir -p "$2"27 28OUT=$(realpath "$1")29MNT=$(realpath "$2")30 31rm -f "$OUT/*.log"32rm -f "$OUT/*.exit"33rm -f "$OUT/*.md"34 35sd=`dirname $0`36cd $sd/../37SRC=`pwd`38 39CMAKE_EXTRA="-DLLAMA_FATAL_WARNINGS=ON"40 41if [ ! -z ${GG_BUILD_METAL} ]; then42    CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_METAL=ON -DGGML_METAL_USE_BF16=ON"43fi44 45if [ ! -z ${GG_BUILD_CUDA} ]; then46    CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_CUDA=ON -DCMAKE_CUDA_ARCHITECTURES=native"47fi48 49if [ ! -z ${GG_BUILD_SYCL} ]; then50    if [ -z ${ONEAPI_ROOT} ]; then51        echo "Not detected ONEAPI_ROOT, please install oneAPI base toolkit and enable it by:"52        echo "source /opt/intel/oneapi/setvars.sh"53        exit 154    fi55 56    CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_SYCL=1 -DCMAKE_C_COMPILER=icx -DCMAKE_CXX_COMPILER=icpx -DGGML_SYCL_F16=ON"57fi58 59if [ ! -z ${GG_BUILD_VULKAN} ]; then60    CMAKE_EXTRA="${CMAKE_EXTRA} -DGGML_VULKAN=1"61fi62## helpers63 64# download a file if it does not exist or if it is outdated65function gg_wget {66    local out=$167    local url=$268 69    local cwd=`pwd`70 71    mkdir -p $out72    cd $out73 74    # should not re-download if file is the same75    wget -nv -N $url76 77    cd $cwd78}79 80function gg_printf {81    printf -- "$@" >> $OUT/README.md82}83 84function gg_run {85    ci=$186 87    set -o pipefail88    set -x89 90    gg_run_$ci | tee $OUT/$ci.log91    cur=$?92    echo "$cur" > $OUT/$ci.exit93 94    set +x95    set +o pipefail96 97    gg_sum_$ci98 99    ret=$((ret | cur))100}101 102## ci103 104# ctest_debug105 106function gg_run_ctest_debug {107    cd ${SRC}108 109    rm -rf build-ci-debug && mkdir build-ci-debug && cd build-ci-debug110 111    set -e112 113    # Check cmake, make and ctest are installed114    gg_check_build_requirements115 116    (time cmake -DCMAKE_BUILD_TYPE=Debug ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log117    (time make -j$(nproc)                                  ) 2>&1 | tee -a $OUT/${ci}-make.log118 119    (time ctest --output-on-failure -L main -E test-opt ) 2>&1 | tee -a $OUT/${ci}-ctest.log120 121    set +e122}123 124function gg_sum_ctest_debug {125    gg_printf '### %s\n\n' "${ci}"126 127    gg_printf 'Runs ctest in debug mode\n'128    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"129    gg_printf '```\n'130    gg_printf '%s\n' "$(cat $OUT/${ci}-ctest.log)"131    gg_printf '```\n'132    gg_printf '\n'133}134 135# ctest_release136 137function gg_run_ctest_release {138    cd ${SRC}139 140    rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release141 142    set -e143 144    # Check cmake, make and ctest are installed145    gg_check_build_requirements146 147    (time cmake -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log148    (time make -j$(nproc)                                    ) 2>&1 | tee -a $OUT/${ci}-make.log149 150    if [ -z ${GG_BUILD_LOW_PERF} ]; then151        (time ctest --output-on-failure -L main ) 2>&1 | tee -a $OUT/${ci}-ctest.log152    else153        (time ctest --output-on-failure -L main -E test-opt ) 2>&1 | tee -a $OUT/${ci}-ctest.log154    fi155 156    set +e157}158 159function gg_sum_ctest_release {160    gg_printf '### %s\n\n' "${ci}"161 162    gg_printf 'Runs ctest in release mode\n'163    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"164    gg_printf '```\n'165    gg_printf '%s\n' "$(cat $OUT/${ci}-ctest.log)"166    gg_printf '```\n'167}168 169# test_scripts_debug170 171function gg_run_test_scripts_debug {172    cd ${SRC}173 174    set -e175 176    (cd ./examples/gguf-split && time bash tests.sh "$SRC/build-ci-debug/bin" "$MNT/models") 2>&1 | tee -a $OUT/${ci}-scripts.log177    (cd ./examples/quantize   && time bash tests.sh "$SRC/build-ci-debug/bin" "$MNT/models") 2>&1 | tee -a $OUT/${ci}-scripts.log178 179    set +e180}181 182function gg_sum_test_scripts_debug {183    gg_printf '### %s\n\n' "${ci}"184 185    gg_printf 'Runs test scripts in debug mode\n'186    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"187    gg_printf '```\n'188    gg_printf '%s\n' "$(cat $OUT/${ci}-scripts.log)"189    gg_printf '```\n'190    gg_printf '\n'191}192 193# test_scripts_release194 195function gg_run_test_scripts_release {196    cd ${SRC}197 198    set -e199 200    (cd ./examples/gguf-split && time bash tests.sh "$SRC/build-ci-release/bin" "$MNT/models") 2>&1 | tee -a $OUT/${ci}-scripts.log201    (cd ./examples/quantize   && time bash tests.sh "$SRC/build-ci-release/bin" "$MNT/models") 2>&1 | tee -a $OUT/${ci}-scripts.log202 203    set +e204}205 206function gg_sum_test_scripts_release {207    gg_printf '### %s\n\n' "${ci}"208 209    gg_printf 'Runs test scripts in release mode\n'210    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"211    gg_printf '```\n'212    gg_printf '%s\n' "$(cat $OUT/${ci}-scripts.log)"213    gg_printf '```\n'214    gg_printf '\n'215}216 217function gg_get_model {218    local gguf_0="$MNT/models/pythia/1.4B/ggml-model-f16.gguf"219    local gguf_1="$MNT/models/pythia/2.8B/ggml-model-f16.gguf"220    local gguf_2="$MNT/models/open-llama/7B-v2/ggml-model-f16.gguf"221    if [[ -s $gguf_0 ]]; then222        echo -n "$gguf_0"223    elif [[ -s $gguf_1 ]]; then224        echo -n "$gguf_1"225    elif [[ -s $gguf_2 ]]; then226        echo -n "$gguf_2"227    else228        echo >&2 "No model found. Can't run gg_run_ctest_with_model."229        exit 1230    fi231}232 233function gg_run_ctest_with_model_debug {234    cd ${SRC}235 236    local model; model=$(gg_get_model)237    cd build-ci-debug238    set -e239    (LLAMACPP_TEST_MODELFILE="$model" time ctest --output-on-failure -L model) 2>&1 | tee -a $OUT/${ci}-ctest.log240    set +e241    cd ..242}243 244function gg_run_ctest_with_model_release {245    cd ${SRC}246 247    local model; model=$(gg_get_model)248    cd build-ci-release249    set -e250    (LLAMACPP_TEST_MODELFILE="$model" time ctest --output-on-failure -L model) 2>&1 | tee -a $OUT/${ci}-ctest.log251    set +e252    cd ..253}254 255function gg_sum_ctest_with_model_debug {256    gg_printf '### %s\n\n' "${ci}"257 258    gg_printf 'Runs ctest with model files in debug mode\n'259    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"260    gg_printf '```\n'261    gg_printf '%s\n' "$(cat $OUT/${ci}-ctest.log)"262    gg_printf '```\n'263}264 265function gg_sum_ctest_with_model_release {266    gg_printf '### %s\n\n' "${ci}"267 268    gg_printf 'Runs ctest with model files in release mode\n'269    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"270    gg_printf '```\n'271    gg_printf '%s\n' "$(cat $OUT/${ci}-ctest.log)"272    gg_printf '```\n'273}274 275# open_llama_7b_v2276 277function gg_run_open_llama_7b_v2 {278    cd ${SRC}279 280    gg_wget models-mnt/open-llama/7B-v2/ https://huggingface.co/openlm-research/open_llama_7b_v2/raw/main/config.json281    gg_wget models-mnt/open-llama/7B-v2/ https://huggingface.co/openlm-research/open_llama_7b_v2/resolve/main/tokenizer.model282    gg_wget models-mnt/open-llama/7B-v2/ https://huggingface.co/openlm-research/open_llama_7b_v2/raw/main/tokenizer_config.json283    gg_wget models-mnt/open-llama/7B-v2/ https://huggingface.co/openlm-research/open_llama_7b_v2/raw/main/special_tokens_map.json284    gg_wget models-mnt/open-llama/7B-v2/ https://huggingface.co/openlm-research/open_llama_7b_v2/raw/main/pytorch_model.bin.index.json285    gg_wget models-mnt/open-llama/7B-v2/ https://huggingface.co/openlm-research/open_llama_7b_v2/resolve/main/pytorch_model-00001-of-00002.bin286    gg_wget models-mnt/open-llama/7B-v2/ https://huggingface.co/openlm-research/open_llama_7b_v2/resolve/main/pytorch_model-00002-of-00002.bin287    gg_wget models-mnt/open-llama/7B-v2/ https://huggingface.co/openlm-research/open_llama_7b_v2/raw/main/generation_config.json288 289    gg_wget models-mnt/wikitext/ https://huggingface.co/datasets/ggml-org/ci/resolve/main/wikitext-2-raw-v1.zip290    unzip -o models-mnt/wikitext/wikitext-2-raw-v1.zip -d models-mnt/wikitext/291 292    path_models="../models-mnt/open-llama/7B-v2"293    path_wiki="../models-mnt/wikitext/wikitext-2-raw"294 295    rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release296 297    set -e298 299    (time cmake -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log300    (time make -j$(nproc)                                    ) 2>&1 | tee -a $OUT/${ci}-make.log301 302    python3 ../examples/convert_legacy_llama.py ${path_models} --outfile ${path_models}/ggml-model-f16.gguf303 304    model_f16="${path_models}/ggml-model-f16.gguf"305    model_q8_0="${path_models}/ggml-model-q8_0.gguf"306    model_q4_0="${path_models}/ggml-model-q4_0.gguf"307    model_q4_1="${path_models}/ggml-model-q4_1.gguf"308    model_q5_0="${path_models}/ggml-model-q5_0.gguf"309    model_q5_1="${path_models}/ggml-model-q5_1.gguf"310    model_q2_k="${path_models}/ggml-model-q2_k.gguf"311    model_q3_k="${path_models}/ggml-model-q3_k.gguf"312    model_q4_k="${path_models}/ggml-model-q4_k.gguf"313    model_q5_k="${path_models}/ggml-model-q5_k.gguf"314    model_q6_k="${path_models}/ggml-model-q6_k.gguf"315 316    wiki_test="${path_wiki}/wiki.test.raw"317 318    ./bin/llama-quantize ${model_f16} ${model_q8_0} q8_0319    ./bin/llama-quantize ${model_f16} ${model_q4_0} q4_0320    ./bin/llama-quantize ${model_f16} ${model_q4_1} q4_1321    ./bin/llama-quantize ${model_f16} ${model_q5_0} q5_0322    ./bin/llama-quantize ${model_f16} ${model_q5_1} q5_1323    ./bin/llama-quantize ${model_f16} ${model_q2_k} q2_k324    ./bin/llama-quantize ${model_f16} ${model_q3_k} q3_k325    ./bin/llama-quantize ${model_f16} ${model_q4_k} q4_k326    ./bin/llama-quantize ${model_f16} ${model_q5_k} q5_k327    ./bin/llama-quantize ${model_f16} ${model_q6_k} q6_k328 329    (time ./bin/llama-cli -no-cnv --model ${model_f16}  -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log330    (time ./bin/llama-cli -no-cnv --model ${model_q8_0} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log331    (time ./bin/llama-cli -no-cnv --model ${model_q4_0} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_0.log332    (time ./bin/llama-cli -no-cnv --model ${model_q4_1} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_1.log333    (time ./bin/llama-cli -no-cnv --model ${model_q5_0} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_0.log334    (time ./bin/llama-cli -no-cnv --model ${model_q5_1} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_1.log335    (time ./bin/llama-cli -no-cnv --model ${model_q2_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q2_k.log336    (time ./bin/llama-cli -no-cnv --model ${model_q3_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q3_k.log337    (time ./bin/llama-cli -no-cnv --model ${model_q4_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_k.log338    (time ./bin/llama-cli -no-cnv --model ${model_q5_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_k.log339    (time ./bin/llama-cli -no-cnv --model ${model_q6_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q6_k.log340 341    (time ./bin/llama-perplexity --model ${model_f16}  -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log342    (time ./bin/llama-perplexity --model ${model_q8_0} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log343    (time ./bin/llama-perplexity --model ${model_q4_0} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_0.log344    (time ./bin/llama-perplexity --model ${model_q4_1} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_1.log345    (time ./bin/llama-perplexity --model ${model_q5_0} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_0.log346    (time ./bin/llama-perplexity --model ${model_q5_1} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_1.log347    (time ./bin/llama-perplexity --model ${model_q2_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q2_k.log348    (time ./bin/llama-perplexity --model ${model_q3_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q3_k.log349    (time ./bin/llama-perplexity --model ${model_q4_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_k.log350    (time ./bin/llama-perplexity --model ${model_q5_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_k.log351    (time ./bin/llama-perplexity --model ${model_q6_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q6_k.log352 353    (time ./bin/llama-imatrix --model ${model_f16} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-imatrix.log354 355    (time ./bin/llama-save-load-state--model ${model_q4_0} -ngl 10 -c 0     ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log356    (time ./bin/llama-save-load-state--model ${model_q4_0} -ngl 10 -c 0 -fa ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log357    (time ./bin/llama-save-load-state--model ${model_q4_0} -ngl 99 -c 0     ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log358    (time ./bin/llama-save-load-state--model ${model_q4_0} -ngl 99 -c 0 -fa ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log359 360    function check_ppl {361        qnt="$1"362        ppl=$(echo "$2" | grep -oE "[0-9]+\.[0-9]+" | tail -n 1)363 364        if [ $(echo "$ppl > 20.0" | bc) -eq 1 ]; then365            printf '  - %s @ %s (FAIL: ppl > 20.0)\n' "$qnt" "$ppl"366            return 20367        fi368 369        printf '  - %s @ %s OK\n' "$qnt" "$ppl"370        return 0371    }372 373    check_ppl "f16"  "$(cat $OUT/${ci}-tg-f16.log  | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log374    check_ppl "q8_0" "$(cat $OUT/${ci}-tg-q8_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log375    check_ppl "q4_0" "$(cat $OUT/${ci}-tg-q4_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log376    check_ppl "q4_1" "$(cat $OUT/${ci}-tg-q4_1.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log377    check_ppl "q5_0" "$(cat $OUT/${ci}-tg-q5_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log378    check_ppl "q5_1" "$(cat $OUT/${ci}-tg-q5_1.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log379    check_ppl "q2_k" "$(cat $OUT/${ci}-tg-q2_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log380    check_ppl "q3_k" "$(cat $OUT/${ci}-tg-q3_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log381    check_ppl "q4_k" "$(cat $OUT/${ci}-tg-q4_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log382    check_ppl "q5_k" "$(cat $OUT/${ci}-tg-q5_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log383    check_ppl "q6_k" "$(cat $OUT/${ci}-tg-q6_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log384 385    cat $OUT/${ci}-imatrix.log | grep "Final" >> $OUT/${ci}-imatrix-sum.log386 387    set +e388}389 390function gg_sum_open_llama_7b_v2 {391    gg_printf '### %s\n\n' "${ci}"392 393    gg_printf 'OpenLLaMA 7B-v2:\n'394    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"395    gg_printf '- perplexity:\n%s\n' "$(cat $OUT/${ci}-ppl.log)"396    gg_printf '- imatrix:\n```\n%s\n```\n' "$(cat $OUT/${ci}-imatrix-sum.log)"397    gg_printf '- f16: \n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-f16.log)"398    gg_printf '- q8_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q8_0.log)"399    gg_printf '- q4_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_0.log)"400    gg_printf '- q4_1:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_1.log)"401    gg_printf '- q5_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_0.log)"402    gg_printf '- q5_1:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_1.log)"403    gg_printf '- q2_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q2_k.log)"404    gg_printf '- q3_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q3_k.log)"405    gg_printf '- q4_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_k.log)"406    gg_printf '- q5_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_k.log)"407    gg_printf '- q6_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q6_k.log)"408    gg_printf '- save-load-state: \n```\n%s\n```\n' "$(cat $OUT/${ci}-save-load-state.log)"409}410 411# pythia_1.4b412 413function gg_run_pythia_1_4b {414    cd ${SRC}415 416    gg_wget models-mnt/pythia/1.4B/ https://huggingface.co/EleutherAI/pythia-1.4b/raw/main/config.json417    gg_wget models-mnt/pythia/1.4B/ https://huggingface.co/EleutherAI/pythia-1.4b/raw/main/tokenizer.json418    gg_wget models-mnt/pythia/1.4B/ https://huggingface.co/EleutherAI/pythia-1.4b/raw/main/tokenizer_config.json419    gg_wget models-mnt/pythia/1.4B/ https://huggingface.co/EleutherAI/pythia-1.4b/raw/main/special_tokens_map.json420    gg_wget models-mnt/pythia/1.4B/ https://huggingface.co/EleutherAI/pythia-1.4b/resolve/main/pytorch_model.bin421 422    gg_wget models-mnt/wikitext/ https://huggingface.co/datasets/ggml-org/ci/resolve/main/wikitext-2-raw-v1.zip423    unzip -o models-mnt/wikitext/wikitext-2-raw-v1.zip -d models-mnt/wikitext/424    head -n 60 models-mnt/wikitext/wikitext-2-raw/wiki.test.raw > models-mnt/wikitext/wikitext-2-raw/wiki.test-60.raw425 426    path_models="../models-mnt/pythia/1.4B"427    path_wiki="../models-mnt/wikitext/wikitext-2-raw"428 429    rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release430 431    set -e432 433    (time cmake -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log434    (time make -j$(nproc)                                    ) 2>&1 | tee -a $OUT/${ci}-make.log435 436    python3 ../convert_hf_to_gguf.py ${path_models} --outfile ${path_models}/ggml-model-f16.gguf437 438    model_f16="${path_models}/ggml-model-f16.gguf"439    model_q8_0="${path_models}/ggml-model-q8_0.gguf"440    model_q4_0="${path_models}/ggml-model-q4_0.gguf"441    model_q4_1="${path_models}/ggml-model-q4_1.gguf"442    model_q5_0="${path_models}/ggml-model-q5_0.gguf"443    model_q5_1="${path_models}/ggml-model-q5_1.gguf"444    model_q2_k="${path_models}/ggml-model-q2_k.gguf"445    model_q3_k="${path_models}/ggml-model-q3_k.gguf"446    model_q4_k="${path_models}/ggml-model-q4_k.gguf"447    model_q5_k="${path_models}/ggml-model-q5_k.gguf"448    model_q6_k="${path_models}/ggml-model-q6_k.gguf"449 450    wiki_test_60="${path_wiki}/wiki.test-60.raw"451 452    ./bin/llama-quantize ${model_f16} ${model_q8_0} q8_0453    ./bin/llama-quantize ${model_f16} ${model_q4_0} q4_0454    ./bin/llama-quantize ${model_f16} ${model_q4_1} q4_1455    ./bin/llama-quantize ${model_f16} ${model_q5_0} q5_0456    ./bin/llama-quantize ${model_f16} ${model_q5_1} q5_1457    ./bin/llama-quantize ${model_f16} ${model_q2_k} q2_k458    ./bin/llama-quantize ${model_f16} ${model_q3_k} q3_k459    ./bin/llama-quantize ${model_f16} ${model_q4_k} q4_k460    ./bin/llama-quantize ${model_f16} ${model_q5_k} q5_k461    ./bin/llama-quantize ${model_f16} ${model_q6_k} q6_k462 463    (time ./bin/llama-cli -no-cnv --model ${model_f16}  -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log464    (time ./bin/llama-cli -no-cnv --model ${model_q8_0} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log465    (time ./bin/llama-cli -no-cnv --model ${model_q4_0} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_0.log466    (time ./bin/llama-cli -no-cnv --model ${model_q4_1} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_1.log467    (time ./bin/llama-cli -no-cnv --model ${model_q5_0} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_0.log468    (time ./bin/llama-cli -no-cnv --model ${model_q5_1} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_1.log469    (time ./bin/llama-cli -no-cnv --model ${model_q2_k} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q2_k.log470    (time ./bin/llama-cli -no-cnv --model ${model_q3_k} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q3_k.log471    (time ./bin/llama-cli -no-cnv --model ${model_q4_k} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_k.log472    (time ./bin/llama-cli -no-cnv --model ${model_q5_k} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_k.log473    (time ./bin/llama-cli -no-cnv --model ${model_q6_k} -ngl 99 -c 0 -s 1234 -n 64 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q6_k.log474 475    (time ./bin/llama-perplexity --model ${model_f16}  -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log476    (time ./bin/llama-perplexity --model ${model_q8_0} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log477    (time ./bin/llama-perplexity --model ${model_q4_0} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_0.log478    (time ./bin/llama-perplexity --model ${model_q4_1} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_1.log479    (time ./bin/llama-perplexity --model ${model_q5_0} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_0.log480    (time ./bin/llama-perplexity --model ${model_q5_1} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_1.log481    (time ./bin/llama-perplexity --model ${model_q2_k} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q2_k.log482    (time ./bin/llama-perplexity --model ${model_q3_k} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q3_k.log483    (time ./bin/llama-perplexity --model ${model_q4_k} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_k.log484    (time ./bin/llama-perplexity --model ${model_q5_k} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_k.log485    (time ./bin/llama-perplexity --model ${model_q6_k} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-tg-q6_k.log486 487    (time ./bin/llama-imatrix --model ${model_f16} -f ${wiki_test_60} -ngl 99 -c 128 -b 128 --chunks 1 ) 2>&1 | tee -a $OUT/${ci}-imatrix.log488 489    (time ./bin/llama-save-load-state --model ${model_q4_0} -ngl 99 -c 0     ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log490    (time ./bin/llama-save-load-state --model ${model_q4_0} -ngl 99 -c 0 -fa ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log491 492    function check_ppl {493        qnt="$1"494        ppl=$(echo "$2" | grep -oE "[0-9]+\.[0-9]+" | tail -n 1)495 496        if [ $(echo "$ppl > 20.0" | bc) -eq 1 ]; then497            printf '  - %s @ %s (FAIL: ppl > 20.0)\n' "$qnt" "$ppl"498            return 20499        fi500 501        printf '  - %s @ %s OK\n' "$qnt" "$ppl"502        return 0503    }504 505    check_ppl "f16"  "$(cat $OUT/${ci}-tg-f16.log  | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log506    check_ppl "q8_0" "$(cat $OUT/${ci}-tg-q8_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log507    check_ppl "q4_0" "$(cat $OUT/${ci}-tg-q4_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log508    check_ppl "q4_1" "$(cat $OUT/${ci}-tg-q4_1.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log509    check_ppl "q5_0" "$(cat $OUT/${ci}-tg-q5_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log510    check_ppl "q5_1" "$(cat $OUT/${ci}-tg-q5_1.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log511   #check_ppl "q2_k" "$(cat $OUT/${ci}-tg-q2_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log # note: ppl > 20.0 for this quant and model512    check_ppl "q3_k" "$(cat $OUT/${ci}-tg-q3_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log513    check_ppl "q4_k" "$(cat $OUT/${ci}-tg-q4_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log514    check_ppl "q5_k" "$(cat $OUT/${ci}-tg-q5_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log515    check_ppl "q6_k" "$(cat $OUT/${ci}-tg-q6_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log516 517    cat $OUT/${ci}-imatrix.log | grep "Final" >> $OUT/${ci}-imatrix-sum.log518 519    set +e520}521 522function gg_sum_pythia_1_4b {523    gg_printf '### %s\n\n' "${ci}"524 525    gg_printf 'Pythia 1.4B:\n'526    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"527    gg_printf '- perplexity:\n%s\n' "$(cat $OUT/${ci}-ppl.log)"528    gg_printf '- imatrix:\n```\n%s\n```\n' "$(cat $OUT/${ci}-imatrix-sum.log)"529    gg_printf '- f16: \n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-f16.log)"530    gg_printf '- q8_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q8_0.log)"531    gg_printf '- q4_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_0.log)"532    gg_printf '- q4_1:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_1.log)"533    gg_printf '- q5_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_0.log)"534    gg_printf '- q5_1:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_1.log)"535    gg_printf '- q2_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q2_k.log)"536    gg_printf '- q3_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q3_k.log)"537    gg_printf '- q4_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_k.log)"538    gg_printf '- q5_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_k.log)"539    gg_printf '- q6_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q6_k.log)"540    gg_printf '- save-load-state: \n```\n%s\n```\n' "$(cat $OUT/${ci}-save-load-state.log)"541}542 543# pythia_2_8b544 545function gg_run_pythia_2_8b {546    cd ${SRC}547 548    gg_wget models-mnt/pythia/2.8B/ https://huggingface.co/EleutherAI/pythia-2.8b/raw/main/config.json549    gg_wget models-mnt/pythia/2.8B/ https://huggingface.co/EleutherAI/pythia-2.8b/raw/main/tokenizer.json550    gg_wget models-mnt/pythia/2.8B/ https://huggingface.co/EleutherAI/pythia-2.8b/raw/main/tokenizer_config.json551    gg_wget models-mnt/pythia/2.8B/ https://huggingface.co/EleutherAI/pythia-2.8b/raw/main/special_tokens_map.json552    gg_wget models-mnt/pythia/2.8B/ https://huggingface.co/EleutherAI/pythia-2.8b/resolve/main/pytorch_model.bin553 554    gg_wget models-mnt/wikitext/ https://huggingface.co/datasets/ggml-org/ci/resolve/main/wikitext-2-raw-v1.zip555    unzip -o models-mnt/wikitext/wikitext-2-raw-v1.zip -d models-mnt/wikitext/556 557    path_models="../models-mnt/pythia/2.8B"558    path_wiki="../models-mnt/wikitext/wikitext-2-raw"559 560    rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release561 562    set -e563 564    (time cmake -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log565    (time make -j$(nproc)                                    ) 2>&1 | tee -a $OUT/${ci}-make.log566 567    python3 ../convert_hf_to_gguf.py ${path_models} --outfile ${path_models}/ggml-model-f16.gguf568 569    model_f16="${path_models}/ggml-model-f16.gguf"570    model_q8_0="${path_models}/ggml-model-q8_0.gguf"571    model_q4_0="${path_models}/ggml-model-q4_0.gguf"572    model_q4_1="${path_models}/ggml-model-q4_1.gguf"573    model_q5_0="${path_models}/ggml-model-q5_0.gguf"574    model_q5_1="${path_models}/ggml-model-q5_1.gguf"575    model_q2_k="${path_models}/ggml-model-q2_k.gguf"576    model_q3_k="${path_models}/ggml-model-q3_k.gguf"577    model_q4_k="${path_models}/ggml-model-q4_k.gguf"578    model_q5_k="${path_models}/ggml-model-q5_k.gguf"579    model_q6_k="${path_models}/ggml-model-q6_k.gguf"580 581    wiki_test="${path_wiki}/wiki.test.raw"582 583    ./bin/llama-quantize ${model_f16} ${model_q8_0} q8_0584    ./bin/llama-quantize ${model_f16} ${model_q4_0} q4_0585    ./bin/llama-quantize ${model_f16} ${model_q4_1} q4_1586    ./bin/llama-quantize ${model_f16} ${model_q5_0} q5_0587    ./bin/llama-quantize ${model_f16} ${model_q5_1} q5_1588    ./bin/llama-quantize ${model_f16} ${model_q2_k} q2_k589    ./bin/llama-quantize ${model_f16} ${model_q3_k} q3_k590    ./bin/llama-quantize ${model_f16} ${model_q4_k} q4_k591    ./bin/llama-quantize ${model_f16} ${model_q5_k} q5_k592    ./bin/llama-quantize ${model_f16} ${model_q6_k} q6_k593 594    (time ./bin/llama-cli -no-cnv --model ${model_f16}  -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log595    (time ./bin/llama-cli -no-cnv --model ${model_q8_0} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log596    (time ./bin/llama-cli -no-cnv --model ${model_q4_0} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_0.log597    (time ./bin/llama-cli -no-cnv --model ${model_q4_1} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_1.log598    (time ./bin/llama-cli -no-cnv --model ${model_q5_0} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_0.log599    (time ./bin/llama-cli -no-cnv --model ${model_q5_1} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_1.log600    (time ./bin/llama-cli -no-cnv --model ${model_q2_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q2_k.log601    (time ./bin/llama-cli -no-cnv --model ${model_q3_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q3_k.log602    (time ./bin/llama-cli -no-cnv --model ${model_q4_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q4_k.log603    (time ./bin/llama-cli -no-cnv --model ${model_q5_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q5_k.log604    (time ./bin/llama-cli -no-cnv --model ${model_q6_k} -t 1 -ngl 99 -c 0 -s 1234 -n 256 --ignore-eos -p "I believe the meaning of life is" ) 2>&1 | tee -a $OUT/${ci}-tg-q6_k.log605 606    (time ./bin/llama-perplexity --model ${model_f16}  -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log607    (time ./bin/llama-perplexity --model ${model_q8_0} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log608    (time ./bin/llama-perplexity --model ${model_q4_0} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_0.log609    (time ./bin/llama-perplexity --model ${model_q4_1} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_1.log610    (time ./bin/llama-perplexity --model ${model_q5_0} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_0.log611    (time ./bin/llama-perplexity --model ${model_q5_1} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_1.log612    (time ./bin/llama-perplexity --model ${model_q2_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q2_k.log613    (time ./bin/llama-perplexity --model ${model_q3_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q3_k.log614    (time ./bin/llama-perplexity --model ${model_q4_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q4_k.log615    (time ./bin/llama-perplexity --model ${model_q5_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q5_k.log616    (time ./bin/llama-perplexity --model ${model_q6_k} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-tg-q6_k.log617 618    (time ./bin/llama-imatrix --model ${model_f16} -f ${wiki_test} -t 1 -ngl 99 -c 2048 -b 512 --chunks 4 ) 2>&1 | tee -a $OUT/${ci}-imatrix.log619 620    (time ./bin/llama-save-load-state --model ${model_q4_0} -ngl 10 -c 0     ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log621    (time ./bin/llama-save-load-state --model ${model_q4_0} -ngl 10 -c 0 -fa ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log622    (time ./bin/llama-save-load-state --model ${model_q4_0} -ngl 99 -c 0     ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log623    (time ./bin/llama-save-load-state --model ${model_q4_0} -ngl 99 -c 0 -fa ) 2>&1 | tee -a $OUT/${ci}-save-load-state.log624 625    function check_ppl {626        qnt="$1"627        ppl=$(echo "$2" | grep -oE "[0-9]+\.[0-9]+" | tail -n 1)628 629        if [ $(echo "$ppl > 20.0" | bc) -eq 1 ]; then630            printf '  - %s @ %s (FAIL: ppl > 20.0)\n' "$qnt" "$ppl"631            return 20632        fi633 634        printf '  - %s @ %s OK\n' "$qnt" "$ppl"635        return 0636    }637 638    check_ppl "f16"  "$(cat $OUT/${ci}-tg-f16.log  | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log639    check_ppl "q8_0" "$(cat $OUT/${ci}-tg-q8_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log640    check_ppl "q4_0" "$(cat $OUT/${ci}-tg-q4_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log641    check_ppl "q4_1" "$(cat $OUT/${ci}-tg-q4_1.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log642    check_ppl "q5_0" "$(cat $OUT/${ci}-tg-q5_0.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log643    check_ppl "q5_1" "$(cat $OUT/${ci}-tg-q5_1.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log644   #check_ppl "q2_k" "$(cat $OUT/${ci}-tg-q2_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log # note: ppl > 20.0 for this quant and model645    check_ppl "q3_k" "$(cat $OUT/${ci}-tg-q3_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log646    check_ppl "q4_k" "$(cat $OUT/${ci}-tg-q4_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log647    check_ppl "q5_k" "$(cat $OUT/${ci}-tg-q5_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log648    check_ppl "q6_k" "$(cat $OUT/${ci}-tg-q6_k.log | grep "^\[1\]")" | tee -a $OUT/${ci}-ppl.log649 650    cat $OUT/${ci}-imatrix.log | grep "Final" >> $OUT/${ci}-imatrix-sum.log651 652    set +e653}654 655function gg_sum_pythia_2_8b {656    gg_printf '### %s\n\n' "${ci}"657 658    gg_printf 'Pythia 2.8B:\n'659    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"660    gg_printf '- perplexity:\n%s\n' "$(cat $OUT/${ci}-ppl.log)"661    gg_printf '- imatrix:\n```\n%s\n```\n' "$(cat $OUT/${ci}-imatrix-sum.log)"662    gg_printf '- f16: \n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-f16.log)"663    gg_printf '- q8_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q8_0.log)"664    gg_printf '- q4_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_0.log)"665    gg_printf '- q4_1:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_1.log)"666    gg_printf '- q5_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_0.log)"667    gg_printf '- q5_1:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_1.log)"668    gg_printf '- q2_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q2_k.log)"669    gg_printf '- q3_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q3_k.log)"670    gg_printf '- q4_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q4_k.log)"671    gg_printf '- q5_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q5_k.log)"672    gg_printf '- q6_k:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q6_k.log)"673    gg_printf '- save-load-state: \n```\n%s\n```\n' "$(cat $OUT/${ci}-save-load-state.log)"674}675 676# bge-small677 678function gg_run_embd_bge_small {679    cd ${SRC}680 681    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/config.json682    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/tokenizer.json683    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/tokenizer_config.json684    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/special_tokens_map.json685    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/resolve/main/pytorch_model.bin686    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/sentence_bert_config.json687    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/vocab.txt688    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/modules.json689    gg_wget models-mnt/bge-small/ https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/config.json690 691    gg_wget models-mnt/bge-small/1_Pooling https://huggingface.co/BAAI/bge-small-en-v1.5/raw/main/1_Pooling/config.json692 693    path_models="../models-mnt/bge-small"694 695    rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release696 697    set -e698 699    (time cmake -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log700    (time make -j$(nproc)                                    ) 2>&1 | tee -a $OUT/${ci}-make.log701 702    python3 ../convert_hf_to_gguf.py ${path_models} --outfile ${path_models}/ggml-model-f16.gguf703 704    model_f16="${path_models}/ggml-model-f16.gguf"705    model_q8_0="${path_models}/ggml-model-q8_0.gguf"706 707    ./bin/llama-quantize ${model_f16} ${model_q8_0} q8_0708 709    (time ./bin/llama-embedding --model ${model_f16}  -p "I believe the meaning of life is" -ngl 99 -c 0 ) 2>&1 | tee -a $OUT/${ci}-tg-f16.log710    (time ./bin/llama-embedding --model ${model_q8_0} -p "I believe the meaning of life is" -ngl 99 -c 0 ) 2>&1 | tee -a $OUT/${ci}-tg-q8_0.log711 712    set +e713}714 715function gg_sum_embd_bge_small {716    gg_printf '### %s\n\n' "${ci}"717 718    gg_printf 'BGE Small (BERT):\n'719    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"720    gg_printf '- f16: \n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-f16.log)"721    gg_printf '- q8_0:\n```\n%s\n```\n' "$(cat $OUT/${ci}-tg-q8_0.log)"722}723 724# rerank_tiny725 726function gg_run_rerank_tiny {727    cd ${SRC}728 729    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/config.json730    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/tokenizer.json731    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/tokenizer_config.json732    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/special_tokens_map.json733    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/resolve/main/pytorch_model.bin734    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/sentence_bert_config.json735    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/vocab.txt736    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/modules.json737    gg_wget models-mnt/rerank-tiny/ https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/config.json738 739    gg_wget models-mnt/rerank-tiny/1_Pooling https://huggingface.co/jinaai/jina-reranker-v1-tiny-en/raw/main/1_Pooling/config.json740 741    path_models="../models-mnt/rerank-tiny"742 743    rm -rf build-ci-release && mkdir build-ci-release && cd build-ci-release744 745    set -e746 747    (time cmake -DCMAKE_BUILD_TYPE=Release ${CMAKE_EXTRA} .. ) 2>&1 | tee -a $OUT/${ci}-cmake.log748    (time make -j$(nproc)                                    ) 2>&1 | tee -a $OUT/${ci}-make.log749 750    python3 ../convert_hf_to_gguf.py ${path_models} --outfile ${path_models}/ggml-model-f16.gguf751 752    model_f16="${path_models}/ggml-model-f16.gguf"753 754    # for this model, the SEP token is "</s>"755    (time ./bin/llama-embedding --model ${model_f16} -p "what is panda?</s></s>hi\nwhat is panda?</s></s>it's a bear\nwhat is panda?</s></s>The giant panda (Ailuropoda melanoleuca), sometimes called a panda bear or simply panda, is a bear species endemic to China." -ngl 99 -c 0 --pooling rank --embd-normalize -1 --verbose-prompt) 2>&1 | tee -a $OUT/${ci}-rk-f16.log756 757    # sample output758    # rerank score 0:    0.029759    # rerank score 1:    0.029760    # rerank score 2:    0.135761 762    # check that the score is in the range [$3, $4]763    function check_score {764        qnt="$1"765        score=$(echo "$2" | grep -oE "[0-9]+\.[0-9]+" | tail -n 1)766 767        if [ $(echo "$score < $3" | bc) -eq 1 ] || [ $(echo "$score > $4" | bc) -eq 1 ]; then768            printf '  - %s @ %s (FAIL: score not in range [%s, %s])\n' "$qnt" "$score" "$3" "$4"769            return 20770        fi771 772        printf '  - %s @ %s OK\n' "$qnt" "$score"773        return 0774    }775 776    check_score "rerank score 0" "$(cat $OUT/${ci}-rk-f16.log | grep "rerank score 0")" "0.00" "0.05" | tee -a $OUT/${ci}-rk-f16.log777    check_score "rerank score 1" "$(cat $OUT/${ci}-rk-f16.log | grep "rerank score 1")" "0.00" "0.05" | tee -a $OUT/${ci}-rk-f16.log778    check_score "rerank score 2" "$(cat $OUT/${ci}-rk-f16.log | grep "rerank score 2")" "0.10" "0.30" | tee -a $OUT/${ci}-rk-f16.log779 780    set +e781}782 783function gg_sum_rerank_tiny {784    gg_printf '### %s\n\n' "${ci}"785 786    gg_printf 'Rerank Tiny (Jina):\n'787    gg_printf '- status: %s\n' "$(cat $OUT/${ci}.exit)"788    gg_printf '- f16: \n```\n%s\n```\n' "$(cat $OUT/${ci}-rk-f16.log)"789}790 791function gg_check_build_requirements {792    if ! command -v cmake &> /dev/null; then793        gg_printf 'cmake not found, please install'794    fi795 796    if ! command -v make &> /dev/null; then797        gg_printf 'make not found, please install'798    fi799 800    if ! command -v ctest &> /dev/null; then801        gg_printf 'ctest not found, please install'802    fi803}804 805## main806 807export LLAMA_LOG_PREFIX=1808export LLAMA_LOG_TIMESTAMPS=1809 810if [ -z ${GG_BUILD_LOW_PERF} ]; then811    # Create symlink: ./llama.cpp/models-mnt -> $MNT/models/models-mnt812    rm -rf ${SRC}/models-mnt813    mnt_models=${MNT}/models814    mkdir -p ${mnt_models}815    ln -sfn ${mnt_models} ${SRC}/models-mnt816 817    # Create a fresh python3 venv and enter it818    if ! python3 -m venv "$MNT/venv"; then819        echo "Error: Failed to create Python virtual environment at $MNT/venv."820        exit 1821    fi822    source "$MNT/venv/bin/activate"823 824    pip install -r ${SRC}/requirements.txt --disable-pip-version-check825    pip install --editable gguf-py --disable-pip-version-check826fi827 828ret=0829 830test $ret -eq 0 && gg_run ctest_debug831test $ret -eq 0 && gg_run ctest_release832 833if [ -z ${GG_BUILD_LOW_PERF} ]; then834    test $ret -eq 0 && gg_run embd_bge_small835    test $ret -eq 0 && gg_run rerank_tiny836 837    if [ -z ${GG_BUILD_CLOUD} ] || [ ${GG_BUILD_EXTRA_TESTS_0} ]; then838        test $ret -eq 0 && gg_run test_scripts_debug839        test $ret -eq 0 && gg_run test_scripts_release840    fi841 842    if [ -z ${GG_BUILD_VRAM_GB} ] || [ ${GG_BUILD_VRAM_GB} -ge 8 ]; then843        if [ -z ${GG_BUILD_CUDA} ] && [ -z ${GG_BUILD_VULKAN} ]; then844            test $ret -eq 0 && gg_run pythia_1_4b845        else846            test $ret -eq 0 && gg_run pythia_2_8b847            #test $ret -eq 0 && gg_run open_llama_7b_v2848        fi849        test $ret -eq 0 && gg_run ctest_with_model_debug850        test $ret -eq 0 && gg_run ctest_with_model_release851    fi852fi853 854exit $ret855