Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
model.py119 linesDownload Raw Back to tiktoken
1from __future__ import annotations
2
3from .core import Encoding
4from .registry import get_encoding
5
6# TODO: these will likely be replaced by an API endpoint
7MODEL_PREFIX_TO_ENCODING: dict[str, str] = {
8    "o1-": "o200k_base",
9    "o3-": "o200k_base",
10    "o4-mini-": "o200k_base",
11    # chat
12    "gpt-5-": "o200k_base",
13    "gpt-4.5-": "o200k_base",
14    "gpt-4.1-": "o200k_base",
15    "chatgpt-4o-": "o200k_base",
16    "gpt-4o-": "o200k_base",  # e.g., gpt-4o-2024-05-13
17    "gpt-4-": "cl100k_base",  # e.g., gpt-4-0314, etc., plus gpt-4-32k
18    "gpt-3.5-turbo-": "cl100k_base",  # e.g, gpt-3.5-turbo-0301, -0401, etc.
19    "gpt-35-turbo-": "cl100k_base",  # Azure deployment name
20    "gpt-oss-": "o200k_harmony",
21    # fine-tuned
22    "ft:gpt-4o": "o200k_base",
23    "ft:gpt-4": "cl100k_base",
24    "ft:gpt-3.5-turbo": "cl100k_base",
25    "ft:davinci-002": "cl100k_base",
26    "ft:babbage-002": "cl100k_base",
27}
28
29MODEL_TO_ENCODING: dict[str, str] = {
30    # reasoning
31    "o1": "o200k_base",
32    "o3": "o200k_base",
33    "o4-mini": "o200k_base",
34    # chat
35    "gpt-5": "o200k_base",
36    "gpt-4.1": "o200k_base",
37    "gpt-4o": "o200k_base",
38    "gpt-4": "cl100k_base",
39    "gpt-3.5-turbo": "cl100k_base",
40    "gpt-3.5": "cl100k_base",  # Common shorthand
41    "gpt-35-turbo": "cl100k_base",  # Azure deployment name
42    # base
43    "davinci-002": "cl100k_base",
44    "babbage-002": "cl100k_base",
45    # embeddings
46    "text-embedding-ada-002": "cl100k_base",
47    "text-embedding-3-small": "cl100k_base",
48    "text-embedding-3-large": "cl100k_base",
49    # DEPRECATED MODELS
50    # text (DEPRECATED)
51    "text-davinci-003": "p50k_base",
52    "text-davinci-002": "p50k_base",
53    "text-davinci-001": "r50k_base",
54    "text-curie-001": "r50k_base",
55    "text-babbage-001": "r50k_base",
56    "text-ada-001": "r50k_base",
57    "davinci": "r50k_base",
58    "curie": "r50k_base",
59    "babbage": "r50k_base",
60    "ada": "r50k_base",
61    # code (DEPRECATED)
62    "code-davinci-002": "p50k_base",
63    "code-davinci-001": "p50k_base",
64    "code-cushman-002": "p50k_base",
65    "code-cushman-001": "p50k_base",
66    "davinci-codex": "p50k_base",
67    "cushman-codex": "p50k_base",
68    # edit (DEPRECATED)
69    "text-davinci-edit-001": "p50k_edit",
70    "code-davinci-edit-001": "p50k_edit",
71    # old embeddings (DEPRECATED)
72    "text-similarity-davinci-001": "r50k_base",
73    "text-similarity-curie-001": "r50k_base",
74    "text-similarity-babbage-001": "r50k_base",
75    "text-similarity-ada-001": "r50k_base",
76    "text-search-davinci-doc-001": "r50k_base",
77    "text-search-curie-doc-001": "r50k_base",
78    "text-search-babbage-doc-001": "r50k_base",
79    "text-search-ada-doc-001": "r50k_base",
80    "code-search-babbage-code-001": "r50k_base",
81    "code-search-ada-code-001": "r50k_base",
82    # open source
83    "gpt2": "gpt2",
84    "gpt-2": "gpt2",  # Maintains consistency with gpt-4
85}
86
87
88def encoding_name_for_model(model_name: str) -> str:
89    """Returns the name of the encoding used by a model.
90
91    Raises a KeyError if the model name is not recognised.
92    """
93    encoding_name = None
94    if model_name in MODEL_TO_ENCODING:
95        encoding_name = MODEL_TO_ENCODING[model_name]
96    else:
97        # Check if the model matches a known prefix
98        # Prefix matching avoids needing library updates for every model version release
99        # Note that this can match on non-existent models (e.g., gpt-3.5-turbo-FAKE)
100        for model_prefix, model_encoding_name in MODEL_PREFIX_TO_ENCODING.items():
101            if model_name.startswith(model_prefix):
102                return model_encoding_name
103
104    if encoding_name is None:
105        raise KeyError(
106            f"Could not automatically map {model_name} to a tokeniser. "
107            "Please use `tiktoken.get_encoding` to explicitly get the tokeniser you expect."
108        ) from None
109
110    return encoding_name
111
112
113def encoding_for_model(model_name: str) -> Encoding:
114    """Returns the encoding used by a model.
115
116    Raises a KeyError if the model name is not recognised.
117    """
118    return get_encoding(encoding_name_for_model(model_name))
119 
codekingpro/portable-devtools · Team Ai