Team Ai
Apppublic

KBaba7/llama.cpp

sourceHugging Faceapache-2.0updated 2y agoView on Hugging Face
0likes
convert_hf_to_gguf_update.py382 linesDownload Raw Back to llama.cpp
1#!/usr/bin/env python32# -*- coding: utf-8 -*-3 4# This script downloads the tokenizer models of the specified models from Huggingface and5# generates the get_vocab_base_pre() function for convert_hf_to_gguf.py6#7# This is necessary in order to analyze the type of pre-tokenizer used by the model and8# provide the necessary information to llama.cpp via the GGUF header in order to implement9# the same pre-tokenizer.10#11# ref: https://github.com/ggerganov/llama.cpp/pull/692012#13# Instructions:14#15# - Add a new model to the "models" list16# - Run the script with your huggingface token:17#18#   python3 convert_hf_to_gguf_update.py <huggingface_token>19#20# - The convert_hf_to_gguf.py script will have had its get_vocab_base_pre() function updated21# - Update llama.cpp with the new pre-tokenizer if necessary22#23# TODO: generate tokenizer tests for llama.cpp24#25 26import logging27import os28import pathlib29import re30 31import requests32import sys33import json34import shutil35 36from hashlib import sha25637from enum import IntEnum, auto38from transformers import AutoTokenizer39 40logging.basicConfig(level=logging.DEBUG)41logger = logging.getLogger("convert_hf_to_gguf_update")42sess = requests.Session()43 44 45class TOKENIZER_TYPE(IntEnum):46    SPM = auto()47    BPE = auto()48    WPM = auto()49    UGM = auto()50 51 52# TODO: this string has to exercise as much pre-tokenizer functionality as possible53#       will be updated with time - contributions welcome54CHK_TXT = '\n \n\n \n\n\n \t \t\t \t\n  \n   \n    \n     \n🚀 (normal) 😶‍🌫️ (multiple emojis concatenated) ✅ 🦙🦙 3 33 333 3333 33333 333333 3333333 33333333 3.3 3..3 3...3 កាន់តែពិសេសអាច😁 ?我想在apple工作1314151天~ ------======= нещо на Български \'\'\'\'\'\'```````\"\"\"\"......!!!!!!?????? I\'ve been \'told he\'s there, \'RE you sure? \'M not sure I\'ll make it, \'D you like some tea? We\'Ve a\'lL'55 56if len(sys.argv) == 2:57    token = sys.argv[1]58    if not token.startswith("hf_"):59        logger.info("Huggingface token seems invalid")60        logger.info("Usage: python convert_hf_to_gguf_update.py <huggingface_token>")61        sys.exit(1)62else:63    logger.info("Usage: python convert_hf_to_gguf_update.py <huggingface_token>")64    sys.exit(1)65 66# TODO: add models here, base models preferred67models = [68    {"name": "llama-spm",        "tokt": TOKENIZER_TYPE.SPM, "repo": "https://huggingface.co/meta-llama/Llama-2-7b-hf", },69    {"name": "llama-bpe",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/meta-llama/Meta-Llama-3-8B", },70    {"name": "phi-3",            "tokt": TOKENIZER_TYPE.SPM, "repo": "https://huggingface.co/microsoft/Phi-3-mini-4k-instruct", },71    {"name": "deepseek-llm",     "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/deepseek-ai/deepseek-llm-7b-base", },72    {"name": "deepseek-coder",   "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/deepseek-ai/deepseek-coder-6.7b-base", },73    {"name": "falcon",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/tiiuae/falcon-7b", },74    {"name": "bert-bge",         "tokt": TOKENIZER_TYPE.WPM, "repo": "https://huggingface.co/BAAI/bge-small-en-v1.5", },75    {"name": "falcon3",          "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/tiiuae/Falcon3-7B-Base", },76    {"name": "bert-bge-large",   "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/BAAI/bge-large-zh-v1.5", },77    {"name": "mpt",              "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/mosaicml/mpt-7b", },78    {"name": "starcoder",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/bigcode/starcoder2-3b", },79    {"name": "gpt-2",            "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/openai-community/gpt2", },80    {"name": "stablelm2",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/stabilityai/stablelm-2-zephyr-1_6b", },81    {"name": "refact",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/smallcloudai/Refact-1_6-base", },82    {"name": "command-r",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/CohereForAI/c4ai-command-r-v01", },83    {"name": "qwen2",            "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/Qwen/Qwen1.5-7B", },84    {"name": "olmo",             "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/allenai/OLMo-1.7-7B-hf", },85    {"name": "dbrx",             "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/databricks/dbrx-base", },86    {"name": "jina-v1-en",       "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/jinaai/jina-reranker-v1-tiny-en", },87    {"name": "jina-v2-en",       "tokt": TOKENIZER_TYPE.WPM, "repo": "https://huggingface.co/jinaai/jina-embeddings-v2-base-en", }, # WPM!88    {"name": "jina-v2-es",       "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/jinaai/jina-embeddings-v2-base-es", },89    {"name": "jina-v2-de",       "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/jinaai/jina-embeddings-v2-base-de", },90    {"name": "smaug-bpe",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/abacusai/Smaug-Llama-3-70B-Instruct", },91    {"name": "poro-chat",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LumiOpen/Poro-34B-chat", },92    {"name": "jina-v2-code",     "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/jinaai/jina-embeddings-v2-base-code", },93    {"name": "viking",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LumiOpen/Viking-7B", }, # Also used for Viking 13B and 33B94    {"name": "gemma",            "tokt": TOKENIZER_TYPE.SPM, "repo": "https://huggingface.co/google/gemma-2b", },95    {"name": "gemma-2",          "tokt": TOKENIZER_TYPE.SPM, "repo": "https://huggingface.co/google/gemma-2-9b", },96    {"name": "jais",             "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/core42/jais-13b", },97    {"name": "t5",               "tokt": TOKENIZER_TYPE.UGM, "repo": "https://huggingface.co/google-t5/t5-small", },98    {"name": "codeshell",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/WisdomShell/CodeShell-7B", },99    {"name": "tekken",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/mistralai/Mistral-Nemo-Base-2407", },100    {"name": "smollm",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/HuggingFaceTB/SmolLM-135M", },101    {'name': "bloom",            "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/bigscience/bloom", },102    {'name': "gpt3-finnish",     "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/TurkuNLP/gpt3-finnish-small", },103    {"name": "exaone",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/LGAI-EXAONE/EXAONE-3.0-7.8B-Instruct", },104    {"name": "phi-2",            "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/microsoft/phi-2", },105    {"name": "chameleon",        "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/facebook/chameleon-7b", },106    {"name": "minerva-7b",       "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/sapienzanlp/Minerva-7B-base-v1.0", },107    {"name": "roberta-bpe",      "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/sentence-transformers/stsb-roberta-base"},108    {"name": "gigachat",         "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/ai-sage/GigaChat-20B-A3B-instruct"},109    {"name": "megrez",           "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/Infinigence/Megrez-3B-Instruct"},110    {"name": "deepseek-v3",      "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/deepseek-ai/DeepSeek-V3"},111    {"name": "deepseek-r1-qwen", "tokt": TOKENIZER_TYPE.BPE, "repo": "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-1.5B"},112]113 114 115def download_file_with_auth(url, token, save_path):116    headers = {"Authorization": f"Bearer {token}"}117    response = sess.get(url, headers=headers)118    response.raise_for_status()119    os.makedirs(os.path.dirname(save_path), exist_ok=True)120    with open(save_path, 'wb') as downloaded_file:121        downloaded_file.write(response.content)122    logger.info(f"File {save_path} downloaded successfully")123 124 125def download_model(model):126    name = model["name"]127    repo = model["repo"]128    tokt = model["tokt"]129 130    os.makedirs(f"models/tokenizers/{name}", exist_ok=True)131 132    files = ["config.json", "tokenizer.json", "tokenizer_config.json"]133 134    if tokt == TOKENIZER_TYPE.SPM:135        files.append("tokenizer.model")136 137    if tokt == TOKENIZER_TYPE.UGM:138        files.append("spiece.model")139 140    if os.path.isdir(repo):141        # If repo is a path on the file system, copy the directory142        for file in files:143            src_path = os.path.join(repo, file)144            dst_path = f"models/tokenizers/{name}/{file}"145            if os.path.isfile(dst_path):146                logger.info(f"{name}: File {dst_path} already exists - skipping")147                continue148            if os.path.isfile(src_path):149                shutil.copy2(src_path, dst_path)150                logger.info(f"{name}: Copied {src_path} to {dst_path}")151            else:152                logger.warning(f"{name}: Source file {src_path} does not exist")153    else:154        # If repo is a URL, download the files155        for file in files:156            save_path = f"models/tokenizers/{name}/{file}"157            if os.path.isfile(save_path):158                logger.info(f"{name}: File {save_path} already exists - skipping")159                continue160            download_file_with_auth(f"{repo}/resolve/main/{file}", token, save_path)161 162 163for model in models:164    try:165        download_model(model)166    except Exception as e:167        logger.error(f"Failed to download model {model['name']}. Error: {e}")168 169 170# generate the source code for the convert_hf_to_gguf.py:get_vocab_base_pre() function:171 172src_ifs = ""173for model in models:174    name = model["name"]175    tokt = model["tokt"]176 177    if tokt == TOKENIZER_TYPE.SPM or tokt == TOKENIZER_TYPE.UGM:178        continue179 180    # Skip if the tokenizer folder does not exist or there are other download issues previously181    if not os.path.exists(f"models/tokenizers/{name}"):182        logger.warning(f"Directory for tokenizer {name} not found. Skipping...")183        continue184 185    # create the tokenizer186    try:187        if name == "t5":188            tokenizer = AutoTokenizer.from_pretrained(f"models/tokenizers/{name}", use_fast=False)189        else:190            tokenizer = AutoTokenizer.from_pretrained(f"models/tokenizers/{name}")191    except OSError as e:192        logger.error(f"Error loading tokenizer for model {name}. The model may not exist or is not accessible with the provided token. Error: {e}")193        continue  # Skip to the next model if the tokenizer can't be loaded194 195    chktok = tokenizer.encode(CHK_TXT)196    chkhsh = sha256(str(chktok).encode()).hexdigest()197 198    logger.info(f"model: {name}")199    logger.info(f"tokt: {tokt}")200    logger.info(f"repo: {model['repo']}")201    logger.info(f"chktok: {chktok}")202    logger.info(f"chkhsh: {chkhsh}")203 204    # print the "pre_tokenizer" content from the tokenizer.json205    with open(f"models/tokenizers/{name}/tokenizer.json", "r", encoding="utf-8") as f:206        cfg = json.load(f)207        normalizer = cfg["normalizer"]208        logger.info("normalizer: " + json.dumps(normalizer, indent=4))209        pre_tokenizer = cfg["pre_tokenizer"]210        logger.info("pre_tokenizer: " + json.dumps(pre_tokenizer, indent=4))211        if "ignore_merges" in cfg["model"]:212            logger.info("ignore_merges: " + json.dumps(cfg["model"]["ignore_merges"], indent=4))213 214    logger.info("")215 216    src_ifs += f"        if chkhsh == \"{chkhsh}\":\n"217    src_ifs += f"            # ref: {model['repo']}\n"218    src_ifs += f"            res = \"{name}\"\n"219 220src_func = f"""221    def get_vocab_base_pre(self, tokenizer) -> str:222        # encoding this string and hashing the resulting tokens would (hopefully) give us a unique identifier that223        # is specific for the BPE pre-tokenizer used by the model224        # we will use this unique identifier to write a "tokenizer.ggml.pre" entry in the GGUF file which we can225        # use in llama.cpp to implement the same pre-tokenizer226 227        chktxt = {repr(CHK_TXT)}228 229        chktok = tokenizer.encode(chktxt)230        chkhsh = sha256(str(chktok).encode()).hexdigest()231 232        logger.debug(f"chktok: {{chktok}}")233        logger.debug(f"chkhsh: {{chkhsh}}")234 235        res = None236 237        # NOTE: if you get an error here, you need to update the convert_hf_to_gguf_update.py script238        #       or pull the latest version of the model from Huggingface239        #       don't edit the hashes manually!240{src_ifs}241        if res is None:242            logger.warning("\\n")243            logger.warning("**************************************************************************************")244            logger.warning("** WARNING: The BPE pre-tokenizer was not recognized!")245            logger.warning("**          There are 2 possible reasons for this:")246            logger.warning("**          - the model has not been added to convert_hf_to_gguf_update.py yet")247            logger.warning("**          - the pre-tokenization config has changed upstream")248            logger.warning("**          Check your model files and convert_hf_to_gguf_update.py and update them accordingly.")249            logger.warning("** ref:     https://github.com/ggerganov/llama.cpp/pull/6920")250            logger.warning("**")251            logger.warning(f"** chkhsh:  {{chkhsh}}")252            logger.warning("**************************************************************************************")253            logger.warning("\\n")254            raise NotImplementedError("BPE pre-tokenizer was not recognized - update get_vocab_base_pre()")255 256        logger.debug(f"tokenizer.ggml.pre: {{repr(res)}}")257        logger.debug(f"chkhsh: {{chkhsh}}")258 259        return res260"""261 262convert_py_pth = pathlib.Path("convert_hf_to_gguf.py")263convert_py = convert_py_pth.read_text(encoding="utf-8")264convert_py = re.sub(265    r"(# Marker: Start get_vocab_base_pre)(.+?)( +# Marker: End get_vocab_base_pre)",266    lambda m: m.group(1) + src_func + m.group(3),267    convert_py,268    flags=re.DOTALL | re.MULTILINE,269)270 271convert_py_pth.write_text(convert_py, encoding="utf-8")272 273logger.info("+++ convert_hf_to_gguf.py was updated")274 275# generate tests for each tokenizer model276 277tests = [278    "ied 4 ½ months",279    "Führer",280    "",281    " ",282    "  ",283    "   ",284    "\t",285    "\n",286    "\n\n",287    "\n\n\n",288    "\t\n",289    "Hello world",290    " Hello world",291    "Hello World",292    " Hello World",293    " Hello World!",294    "Hello, world!",295    " Hello, world!",296    " this is 🦙.cpp",297    "w048 7tuijk dsdfhu",298    "нещо на Български",299    "កាន់តែពិសេសអាចខលចេញ",300    "🚀 (normal) 😶‍🌫️ (multiple emojis concatenated) ✅ (only emoji that has its own token)",301    "Hello",302    " Hello",303    "  Hello",304    "   Hello",305    "    Hello",306    "    Hello\n    Hello",307    " (",308    "\n =",309    "' era",310    "Hello, y'all! How are you 😁 ?我想在apple工作1314151天~",311    "!!!!!!",312    "3",313    "33",314    "333",315    "3333",316    "33333",317    "333333",318    "3333333",319    "33333333",320    "333333333",321    "Cửa Việt", # llama-bpe fails on this322    " discards",323    CHK_TXT,324]325 326# write the tests to ./models/ggml-vocab-{name}.gguf.inp327# the format is:328#329# test0330# __ggml_vocab_test__331# test1332# __ggml_vocab_test__333# ...334#335 336# with each model, encode all tests and write the results in ./models/ggml-vocab-{name}.gguf.out337# for each test, write the resulting tokens on a separate line338 339for model in models:340    name = model["name"]341    tokt = model["tokt"]342 343    # Skip if the tokenizer folder does not exist or there are other download issues previously344    if not os.path.exists(f"models/tokenizers/{name}"):345        logger.warning(f"Directory for tokenizer {name} not found. Skipping...")346        continue347 348    # create the tokenizer349    try:350        if name == "t5":351            tokenizer = AutoTokenizer.from_pretrained(f"models/tokenizers/{name}", use_fast=False)352        else:353            tokenizer = AutoTokenizer.from_pretrained(f"models/tokenizers/{name}")354    except OSError as e:355        logger.error(f"Failed to load tokenizer for model {name}. Error: {e}")356        continue  # Skip this model and continue with the next one in the loop357 358    with open(f"models/ggml-vocab-{name}.gguf.inp", "w", encoding="utf-8") as f:359        for text in tests:360            f.write(f"{text}")361            f.write("\n__ggml_vocab_test__\n")362 363    with open(f"models/ggml-vocab-{name}.gguf.out", "w") as f:364        for text in tests:365            res = tokenizer.encode(text, add_special_tokens=False)366            for r in res:367                f.write(f" {r}")368            f.write("\n")369 370    logger.info(f"Tests for {name} written in ./models/ggml-vocab-{name}.gguf.*")371 372# generate commands for creating vocab files373 374logger.info("\nRun the following commands to generate the vocab files for testing:\n")375 376for model in models:377    name = model["name"]378 379    print(f"python3 convert_hf_to_gguf.py models/tokenizers/{name}/ --outfile models/ggml-vocab-{name}.gguf --vocab-only") # noqa: NP100380 381logger.info("\n")382