codekingpro/portable-devtools
114k
1"""
2Tokenizers Module
3"""
4
5from _typeshed import Incomplete
6from collections.abc import Sequence
7from tokenizers.decoders import Decoder
8from tokenizers.models import Model
9from tokenizers.normalizers import Normalizer
10from tokenizers.pre_tokenizers import PreTokenizer
11from tokenizers.processors import PostProcessor
12from tokenizers.trainers import Trainer
13from typing import Any, Final, final
14
15__version__: Final[str]
16
17@final
18class AddedToken:
19 """
20 Represents a token that can be be added to a :class:`~tokenizers.Tokenizer`.
21 It can have special options that defines the way it should behave.
22
23 Args:
24 content (:obj:`str`): The content of the token
25
26 single_word (:obj:`bool`, defaults to :obj:`False`):
27 Defines whether this token should only match single words. If :obj:`True`, this
28 token will never match inside of a word. For example the token ``ing`` would match
29 on ``tokenizing`` if this option is :obj:`False`, but not if it is :obj:`True`.
30 The notion of "`inside of a word`" is defined by the word boundaries pattern in
31 regular expressions (ie. the token should start and end with word boundaries).
32
33 lstrip (:obj:`bool`, defaults to :obj:`False`):
34 Defines whether this token should strip all potential whitespaces on its left side.
35 If :obj:`True`, this token will greedily match any whitespace on its left. For
36 example if we try to match the token ``[MASK]`` with ``lstrip=True``, in the text
37 ``"I saw a [MASK]"``, we would match on ``" [MASK]"``. (Note the space on the left).
38
39 rstrip (:obj:`bool`, defaults to :obj:`False`):
40 Defines whether this token should strip all potential whitespaces on its right
41 side. If :obj:`True`, this token will greedily match any whitespace on its right.
42 It works just like :obj:`lstrip` but on the right.
43
44 normalized (:obj:`bool`, defaults to :obj:`True` with :meth:`~tokenizers.Tokenizer.add_tokens` and :obj:`False` with :meth:`~tokenizers.Tokenizer.add_special_tokens`):
45 Defines whether this token should match against the normalized version of the input
46 text. For example, with the added token ``"yesterday"``, and a normalizer in charge of
47 lowercasing the text, the token could be extract from the input ``"I saw a lion
48 Yesterday"``.
49 special (:obj:`bool`, defaults to :obj:`False` with :meth:`~tokenizers.Tokenizer.add_tokens` and :obj:`False` with :meth:`~tokenizers.Tokenizer.add_special_tokens`):
50 Defines whether this token should be skipped when decoding.
51 """
52 def __eq__(self, /, other: object) -> bool: ...
53 def __ge__(self, /, other: object) -> bool: ...
54 def __getstate__(self, /) -> dict: ...
55 def __gt__(self, /, other: object) -> bool: ...
56 def __hash__(self, /) -> int: ...
57 def __le__(self, /, other: object) -> bool: ...
58 def __lt__(self, /, other: object) -> bool: ...
59 def __ne__(self, /, other: object) -> bool: ...
60 def __new__(cls, /, content: str | None = None, **kwargs) -> AddedToken: ...
61 def __repr__(self, /) -> str: ...
62 def __setstate__(self, /, state: Any) -> None: ...
63 def __str__(self, /) -> str: ...
64 @property
65 def content(self, /) -> str:
66 """
67 Get the content of this :obj:`AddedToken`
68 """
69 @content.setter
70 def content(self, /, content: str) -> None:
71 """
72 Set the content of this :obj:`AddedToken`
73 """
74 @property
75 def lstrip(self, /) -> bool:
76 """
77 Get the value of the :obj:`lstrip` option
78 """
79 @property
80 def normalized(self, /) -> bool:
81 """
82 Get the value of the :obj:`normalized` option
83 """
84 @property
85 def rstrip(self, /) -> bool:
86 """
87 Get the value of the :obj:`rstrip` option
88 """
89 @property
90 def single_word(self, /) -> bool:
91 """
92 Get the value of the :obj:`single_word` option
93 """
94 @property
95 def special(self, /) -> bool:
96 """
97 Get the value of the :obj:`special` option
98 """
99 @special.setter
100 def special(self, /, special: bool) -> None:
101 """
102 Set the value of the :obj:`special` option
103 """
104
105@final
106class Encoding:
107 """
108 The :class:`~tokenizers.Encoding` represents the output of a :class:`~tokenizers.Tokenizer`.
109
110 It holds all the information about the tokenized input, including the token IDs,
111 token strings, attention masks, offsets, and more. This is the main data structure
112 returned by :meth:`~tokenizers.Tokenizer.encode` and
113 :meth:`~tokenizers.Tokenizer.encode_batch`.
114
115 Example::
116
117 >>> from tokenizers import Tokenizer
118 >>> tokenizer = Tokenizer.from_pretrained("bert-base-uncased")
119 >>> encoding = tokenizer.encode("Hello, world!")
120 >>> encoding.ids
121 [101, 7592, 1010, 2088, 999, 102]
122 >>> encoding.tokens
123 ['[CLS]', 'hello', ',', 'world', '!', '[SEP]']
124 >>> encoding.offsets
125 [(0, 0), (0, 5), (5, 6), (7, 12), (12, 13), (0, 0)]
126 """
127 def __getstate__(self, /) -> Any: ...
128 def __len__(self, /) -> int: ...
129 def __new__(cls, /) -> Encoding: ...
130 def __repr__(self, /) -> str: ...
131 def __setstate__(self, /, state: Any) -> None: ...
132 @property
133 def attention_mask(self, /) -> list[int]:
134 """
135 The attention mask
136
137 This indicates to the LM which tokens should be attended to, and which should not.
138 This is especially important when batching sequences, where we need to applying
139 padding.
140
141 Returns:
142 :obj:`List[int]`: The attention mask
143 """
144 def char_to_token(self, /, char_pos: int, sequence_index: int = 0) -> int | None:
145 """
146 Get the token that contains the char at the given position in the input sequence.
147
148 Args:
149 char_pos (:obj:`int`):
150 The position of a char in the input string
151 sequence_index (:obj:`int`, defaults to :obj:`0`):
152 The index of the sequence that contains the target char
153
154 Returns:
155 :obj:`int`: The index of the token that contains this char in the encoded sequence
156 """
157 def char_to_word(self, /, char_pos: int, sequence_index: int = 0) -> int | None:
158 """
159 Get the word that contains the char at the given position in the input sequence.
160
161 Args:
162 char_pos (:obj:`int`):
163 The position of a char in the input string
164 sequence_index (:obj:`int`, defaults to :obj:`0`):
165 The index of the sequence that contains the target char
166
167 Returns:
168 :obj:`int`: The index of the word that contains this char in the input sequence
169 """
170 @property
171 def ids(self, /) -> list[int]:
172 """
173 The generated IDs
174
175 The IDs are the main input to a Language Model. They are the token indices,
176 the numerical representations that a LM understands.
177
178 Returns:
179 :obj:`List[int]`: The list of IDs
180 """
181 @staticmethod
182 def merge(encodings: Sequence[Encoding], growing_offsets: bool = True) -> "Encoding":
183 """
184 Merge the list of encodings into one final :class:`~tokenizers.Encoding`
185
186 Args:
187 encodings (A :obj:`List` of :class:`~tokenizers.Encoding`):
188 The list of encodings that should be merged in one
189
190 growing_offsets (:obj:`bool`, defaults to :obj:`True`):
191 Whether the offsets should accumulate while merging
192
193 Returns:
194 :class:`~tokenizers.Encoding`: The resulting Encoding
195 """
196 @property
197 def n_sequences(self, /) -> int:
198 """
199 The number of sequences represented
200
201 Returns:
202 :obj:`int`: The number of sequences in this :class:`~tokenizers.Encoding`
203 """
204 @property
205 def offsets(self, /) -> list[tuple[int, int]]:
206 """
207 The offsets associated to each token
208
209 These offsets let's you slice the input string, and thus retrieve the original
210 part that led to producing the corresponding token.
211
212 Returns:
213 A :obj:`List` of :obj:`Tuple[int, int]`: The list of offsets
214 """
215 @property
216 def overflowing(self, /) -> list[Encoding]:
217 """
218 A :obj:`List` of overflowing :class:`~tokenizers.Encoding`
219
220 When using truncation, the :class:`~tokenizers.Tokenizer` takes care of splitting
221 the output into as many pieces as required to match the specified maximum length.
222 This field lets you retrieve all the subsequent pieces.
223
224 When you use pairs of sequences, the overflowing pieces will contain enough
225 variations to cover all the possible combinations, while respecting the provided
226 maximum length.
227 """
228 def pad(self, /, length: int, **kwargs) -> "None":
229 """
230 Pad the :class:`~tokenizers.Encoding` at the given length
231
232 Args:
233 length (:obj:`int`):
234 The desired length
235
236 direction: (:obj:`str`, defaults to :obj:`right`):
237 The expected padding direction. Can be either :obj:`right` or :obj:`left`
238
239 pad_id (:obj:`int`, defaults to :obj:`0`):
240 The ID corresponding to the padding token
241
242 pad_type_id (:obj:`int`, defaults to :obj:`0`):
243 The type ID corresponding to the padding token
244
245 pad_token (:obj:`str`, defaults to `[PAD]`):
246 The pad token to use
247 """
248 @property
249 def sequence_ids(self, /) -> list[int | None]:
250 """
251 The generated sequence indices.
252
253 They represent the index of the input sequence associated to each token.
254 The sequence id can be None if the token is not related to any input sequence,
255 like for example with special tokens.
256
257 Returns:
258 A :obj:`List` of :obj:`Optional[int]`: A list of optional sequence index.
259 """
260 def set_sequence_id(self, /, sequence_id: int) -> None:
261 """
262 Set the given sequence index
263
264 Set the given sequence index for the whole range of tokens contained in this
265 :class:`~tokenizers.Encoding`.
266 """
267 @property
268 def special_tokens_mask(self, /) -> list[int]:
269 """
270 The special token mask
271
272 This indicates which tokens are special tokens, and which are not.
273
274 Returns:
275 :obj:`List[int]`: The special tokens mask
276 """
277 def token_to_chars(self, /, token_index: int) -> tuple[int, int] | None:
278 """
279 Get the offsets of the token at the given index.
280
281 The returned offsets are related to the input sequence that contains the
282 token. In order to determine in which input sequence it belongs, you
283 must call :meth:`~tokenizers.Encoding.token_to_sequence()`.
284
285 Args:
286 token_index (:obj:`int`):
287 The index of a token in the encoded sequence.
288
289 Returns:
290 :obj:`Tuple[int, int]`: The token offsets :obj:`(first, last + 1)`
291 """
292 def token_to_sequence(self, /, token_index: int) -> int | None:
293 """
294 Get the index of the sequence represented by the given token.
295
296 In the general use case, this method returns :obj:`0` for a single sequence or
297 the first sequence of a pair, and :obj:`1` for the second sequence of a pair
298
299 Args:
300 token_index (:obj:`int`):
301 The index of a token in the encoded sequence.
302
303 Returns:
304 :obj:`int`: The sequence id of the given token
305 """
306 def token_to_word(self, /, token_index: int) -> int | None:
307 """
308 Get the index of the word that contains the token in one of the input sequences.
309
310 The returned word index is related to the input sequence that contains
311 the token. In order to determine in which input sequence it belongs, you
312 must call :meth:`~tokenizers.Encoding.token_to_sequence()`.
313
314 Args:
315 token_index (:obj:`int`):
316 The index of a token in the encoded sequence.
317
318 Returns:
319 :obj:`int`: The index of the word in the relevant input sequence.
320 """
321 @property
322 def tokens(self, /) -> list[str]:
323 """
324 The generated tokens
325
326 They are the string representation of the IDs.
327
328 Returns:
329 :obj:`List[str]`: The list of tokens
330 """
331 def truncate(self, /, max_length: int, stride: int = 0, direction: str = "right") -> "None":
332 """
333 Truncate the :class:`~tokenizers.Encoding` at the given length
334
335 If this :class:`~tokenizers.Encoding` represents multiple sequences, when truncating
336 this information is lost. It will be considered as representing a single sequence.
337
338 Args:
339 max_length (:obj:`int`):
340 The desired length
341
342 stride (:obj:`int`, defaults to :obj:`0`):
343 The length of previous content to be included in each overflowing piece
344
345 direction (:obj:`str`, defaults to :obj:`right`):
346 Truncate direction
347 """
348 @property
349 def type_ids(self, /) -> list[int]:
350 """
351 The generated type IDs
352
353 Generally used for tasks like sequence classification or question answering,
354 these tokens let the LM know which input sequence corresponds to each tokens.
355
356 Returns:
357 :obj:`List[int]`: The list of type ids
358 """
359 @property
360 def word_ids(self, /) -> list[int | None]:
361 """
362 The generated word indices.
363
364 They represent the index of the word associated to each token.
365 When the input is pre-tokenized, they correspond to the ID of the given input label,
366 otherwise they correspond to the words indices as defined by the
367 :class:`~tokenizers.pre_tokenizers.PreTokenizer` that was used.
368
369 For special tokens and such (any token that was generated from something that was
370 not part of the input), the output is :obj:`None`
371
372 Returns:
373 A :obj:`List` of :obj:`Optional[int]`: A list of optional word index.
374 """
375 def word_to_chars(self, /, word_index: int, sequence_index: int = 0) -> tuple[int, int] | None:
376 """
377 Get the offsets of the word at the given index in one of the input sequences.
378
379 Args:
380 word_index (:obj:`int`):
381 The index of a word in one of the input sequences.
382 sequence_index (:obj:`int`, defaults to :obj:`0`):
383 The index of the sequence that contains the target word
384
385 Returns:
386 :obj:`Tuple[int, int]`: The range of characters (span) :obj:`(first, last + 1)`
387 """
388 def word_to_tokens(self, /, word_index: int, sequence_index: int = 0) -> tuple[int, int] | None:
389 """
390 Get the encoded tokens corresponding to the word at the given index
391 in one of the input sequences.
392
393 Args:
394 word_index (:obj:`int`):
395 The index of a word in one of the input sequences.
396 sequence_index (:obj:`int`, defaults to :obj:`0`):
397 The index of the sequence that contains the target word
398
399 Returns:
400 :obj:`Tuple[int, int]`: The range of tokens: :obj:`(first, last + 1)`
401 """
402 @property
403 def words(self, /) -> list[int | None]:
404 """
405 The generated word indices.
406
407 .. warning::
408 This is deprecated and will be removed in a future version.
409 Please use :obj:`~tokenizers.Encoding.word_ids` instead.
410
411 They represent the index of the word associated to each token.
412 When the input is pre-tokenized, they correspond to the ID of the given input label,
413 otherwise they correspond to the words indices as defined by the
414 :class:`~tokenizers.pre_tokenizers.PreTokenizer` that was used.
415
416 For special tokens and such (any token that was generated from something that was
417 not part of the input), the output is :obj:`None`
418
419 Returns:
420 A :obj:`List` of :obj:`Optional[int]`: A list of optional word index.
421 """
422
423@final
424class NormalizedString:
425 """
426 NormalizedString
427
428 A NormalizedString takes care of modifying an "original" string, to obtain a "normalized" one.
429 While making all the requested modifications, it keeps track of the alignment information
430 between the two versions of the string.
431
432 Args:
433 sequence: str:
434 The string sequence used to initialize this NormalizedString
435 """
436 def __getitem__(self, /, range: int | tuple[int, int] | slice) -> NormalizedString | None: ...
437 def __new__(cls, /, sequence: str) -> NormalizedString: ...
438 def __repr__(self, /) -> str: ...
439 def __str__(self, /) -> str: ...
440 def append(self, /, s: str) -> None:
441 """
442 Append the given sequence to the string
443 """
444 def clear(self, /) -> None:
445 """
446 Clears the string
447 """
448 def filter(self, /, func: Any) -> None:
449 """
450 Filter each character of the string using the given func
451 """
452 def for_each(self, /, func: Any) -> None:
453 """
454 Calls the given function for each character of the string
455 """
456 def lowercase(self, /) -> None:
457 """
458 Lowercase the string
459 """
460 def lstrip(self, /) -> None:
461 """
462 Strip the left of the string
463 """
464 def map(self, /, func: Any) -> None:
465 """
466 Calls the given function for each character of the string
467
468 Replaces each character of the string using the returned value. Each
469 returned value **must** be a str of length 1 (ie a character).
470 """
471 def nfc(self, /) -> None:
472 """
473 Runs the NFC normalization
474 """
475 def nfd(self, /) -> None:
476 """
477 Runs the NFD normalization
478 """
479 def nfkc(self, /) -> None:
480 """
481 Runs the NFKC normalization
482 """
483 def nfkd(self, /) -> None:
484 """
485 Runs the NFKD normalization
486 """
487 @property
488 def normalized(self, /) -> str:
489 """
490 The normalized part of the string
491 """
492 @property
493 def original(self, /) -> str: ...
494 def prepend(self, /, s: str) -> None:
495 """
496 Prepend the given sequence to the string
497 """
498 def replace(self, /, pattern: str | Regex, content: str) -> None:
499 """
500 Replace the content of the given pattern with the provided content
501
502 Args:
503 pattern: Pattern:
504 A pattern used to match the string. Usually a string or a Regex
505
506 content: str:
507 The content to be used as replacement
508 """
509 def rstrip(self, /) -> None:
510 """
511 Strip the right of the string
512 """
513 def slice(self, /, range: int | tuple[int, int] | slice) -> NormalizedString | None:
514 """
515 Slice the string using the given range
516 """
517 def split(self, /, pattern: str | Regex, behavior: Incomplete) -> list[NormalizedString]:
518 """
519 Split the NormalizedString using the given pattern and the specified behavior
520
521 Args:
522 pattern: Pattern:
523 A pattern used to split the string. Usually a string or a regex built with `tokenizers.Regex`
524
525 behavior: SplitDelimiterBehavior:
526 The behavior to use when splitting.
527 Choices: "removed", "isolated", "merged_with_previous", "merged_with_next",
528 "contiguous"
529
530 Returns:
531 A list of NormalizedString, representing each split
532 """
533 def strip(self, /) -> None:
534 """
535 Strip both ends of the string
536 """
537 def uppercase(self, /) -> None:
538 """
539 Uppercase the string
540 """
541
542@final
543class PreTokenizedString:
544 """
545 PreTokenizedString
546
547 Wrapper over a string, that provides a way to normalize, pre-tokenize, tokenize the
548 underlying string, while keeping track of the alignment information (offsets).
549
550 The PreTokenizedString manages what we call `splits`. Each split represents a substring
551 which is a subpart of the original string, with the relevant offsets and tokens.
552
553 When calling one of the methods used to modify the PreTokenizedString (namely one of
554 `split`, `normalize` or `tokenize), only the `splits` that don't have any associated
555 tokens will get modified.
556
557 Args:
558 sequence: str:
559 The string sequence used to initialize this PreTokenizedString
560 """
561 def __new__(cls, /, s: str) -> PreTokenizedString: ...
562 def get_splits(
563 self, /, offset_referential: Incomplete = ..., offset_type: Incomplete = ...
564 ) -> list[tuple[str, tuple[int, int], list[Token] | None]]:
565 """
566 Get the splits currently managed by the PreTokenizedString
567
568 Args:
569 offset_referential: :obj:`str`
570 Whether the returned splits should have offsets expressed relative
571 to the original string, or the normalized one. choices: "original", "normalized".
572
573 offset_type: :obj:`str`
574 Whether the returned splits should have offsets expressed in bytes or chars.
575 When slicing an str, we usually want to use chars, which is the default value.
576 Now in some cases it might be interesting to get these offsets expressed in bytes,
577 so it is possible to change this here.
578 choices: "char", "bytes"
579
580 Returns
581 A list of splits
582 """
583 def normalize(self, /, func: Any) -> None:
584 """
585 Normalize each split of the `PreTokenizedString` using the given `func`
586
587 Args:
588 func: Callable[[NormalizedString], None]:
589 The function used to normalize each underlying split. This function
590 does not need to return anything, just calling the methods on the provided
591 NormalizedString allow its modification.
592 """
593 def split(self, /, func: Any) -> None:
594 """
595 Split the PreTokenizedString using the given `func`
596
597 Args:
598 func: Callable[[index, NormalizedString], List[NormalizedString]]:
599 The function used to split each underlying split.
600 It is expected to return a list of `NormalizedString`, that represent the new
601 splits. If the given `NormalizedString` does not need any splitting, we can
602 just return it directly.
603 In order for the offsets to be tracked accurately, any returned `NormalizedString`
604 should come from calling either `.split` or `.slice` on the received one.
605 """
606 def to_encoding(self, /, type_id: int = 0, word_idx: int | None = None) -> "Encoding":
607 """
608 Return an Encoding generated from this PreTokenizedString
609
610 Args:
611 type_id: int = 0:
612 The type_id to be used on the generated Encoding.
613
614 word_idx: Optional[int] = None:
615 An optional word index to be used for each token of this Encoding. If provided,
616 all the word indices in the generated Encoding will use this value, instead
617 of the one automatically tracked during pre-tokenization.
618
619 Returns:
620 An Encoding
621 """
622 def tokenize(self, /, func: Any) -> None:
623 """
624 Tokenize each split of the `PreTokenizedString` using the given `func`
625
626 Args:
627 func: Callable[[str], List[Token]]:
628 The function used to tokenize each underlying split. This function must return
629 a list of Token generated from the input str.
630 """
631
632@final
633class Regex:
634 """
635 Instantiate a new Regex with the given pattern
636 """
637 def __new__(cls, /, s: str) -> Regex: ...
638
639@final
640class Token:
641 def __new__(cls, /, id: int, value: str, offsets: tuple[int, int]) -> Token:
642 """
643 Create a token from id, string value and byte offsets
644 """
645 def as_tuple(self, /) -> tuple[int, str, tuple[int, int]]: ...
646 @property
647 def id(self, /) -> int: ...
648 @property
649 def offsets(self, /) -> tuple[int, int]: ...
650 @property
651 def value(self, /) -> str: ...
652
653@final
654class Tokenizer:
655 """
656 A :obj:`Tokenizer` works as a pipeline. It processes some raw text as input
657 and outputs an :class:`~tokenizers.Encoding`.
658
659 The pipeline is structured as follows:
660
661 1. The :class:`~tokenizers.normalizers.Normalizer` normalizes the raw input text.
662 2. The :class:`~tokenizers.pre_tokenizers.PreTokenizer` splits the normalized text
663 into word-level tokens.
664 3. The :class:`~tokenizers.models.Model` tokenizes each word into subword tokens
665 and maps them to IDs.
666 4. The :class:`~tokenizers.processors.PostProcessor` applies any final
667 transformations (e.g., adding special tokens like ``[CLS]`` and ``[SEP]``).
668
669 Args:
670 model (:class:`~tokenizers.models.Model`):
671 The core algorithm that this :obj:`Tokenizer` should be using.
672
673 Example::
674
675 >>> from tokenizers import Tokenizer
676 >>> from tokenizers.models import BPE
677 >>> from tokenizers.normalizers import Lowercase
678 >>> from tokenizers.pre_tokenizers import Whitespace
679 >>> tokenizer = Tokenizer(BPE(unk_token="<unk>"))
680 >>> tokenizer.normalizer = Lowercase()
681 >>> tokenizer.pre_tokenizer = Whitespace()
682 >>> # Load a pre-built tokenizer from HuggingFace Hub
683 >>> tokenizer = Tokenizer.from_pretrained("bert-base-uncased")
684 """
685 def __getnewargs__(self, /) -> tuple: ...
686 def __getstate__(self, /) -> Any: ...
687 def __new__(cls, /, model: Model) -> Tokenizer: ...
688 def __repr__(self, /) -> str: ...
689 def __setstate__(self, /, state: Any) -> None: ...
690 def __str__(self, /) -> str: ...
691 def add_special_tokens(self, /, tokens: list) -> int:
692 """
693 Add the given special tokens to the Tokenizer.
694
695 If these tokens are already part of the vocabulary, it just let the Tokenizer know about
696 them. If they don't exist, the Tokenizer creates them, giving them a new id.
697
698 These special tokens will never be processed by the model (ie won't be split into
699 multiple tokens), and they can be removed from the output when decoding.
700
701 Args:
702 tokens (A :obj:`List` of :class:`~tokenizers.AddedToken` or :obj:`str`):
703 The list of special tokens we want to add to the vocabulary. Each token can either
704 be a string or an instance of :class:`~tokenizers.AddedToken` for more
705 customization.
706
707 Returns:
708 :obj:`int`: The number of tokens that were created in the vocabulary
709 """
710 def add_tokens(self, /, tokens: list) -> int:
711 """
712 Add the given tokens to the vocabulary
713
714 The given tokens are added only if they don't already exist in the vocabulary.
715 Each token then gets a new attributed id.
716
717 Args:
718 tokens (A :obj:`List` of :class:`~tokenizers.AddedToken` or :obj:`str`):
719 The list of tokens we want to add to the vocabulary. Each token can be either a
720 string or an instance of :class:`~tokenizers.AddedToken` for more customization.
721
722 Returns:
723 :obj:`int`: The number of tokens that were created in the vocabulary
724 """
725 def async_decode_batch(self, /, sequences: Sequence[Sequence[int]], skip_special_tokens: bool = True) -> Any:
726 """
727 Decode a batch of ids back to their corresponding string
728
729 Args:
730 sequences (:obj:`List` of :obj:`List[int]`):
731 The batch of sequences we want to decode
732
733 skip_special_tokens (:obj:`bool`, defaults to :obj:`True`):
734 Whether the special tokens should be removed from the decoded strings
735
736 Returns:
737 :obj:`List[str]`: A list of decoded strings
738 """
739 def async_encode(
740 self, /, sequence: Any, pair: Any | None = None, is_pretokenized: bool = False, add_special_tokens: bool = True
741 ) -> Any:
742 """
743 Asynchronously encode the given input with character offsets.
744
745 This is an async version of encode that can be awaited in async Python code.
746
747 Example:
748 Here are some examples of the inputs that are accepted::
749
750 await async_encode("A single sequence")
751
752 Args:
753 sequence (:obj:`~tokenizers.InputSequence`):
754 The main input sequence we want to encode. This sequence can be either raw
755 text or pre-tokenized, according to the ``is_pretokenized`` argument:
756
757 - If ``is_pretokenized=False``: :class:`~tokenizers.TextInputSequence`
758 - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedInputSequence`
759
760 pair (:obj:`~tokenizers.InputSequence`, `optional`):
761 An optional input sequence. The expected format is the same that for ``sequence``.
762
763 is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
764 Whether the input is already pre-tokenized
765
766 add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
767 Whether to add the special tokens
768
769 Returns:
770 :class:`~tokenizers.Encoding`: The encoded result
771 """
772 def async_encode_batch(
773 self, /, input: Sequence[Any], is_pretokenized: bool = False, add_special_tokens: bool = True
774 ) -> Any:
775 """
776 Asynchronously encode the given batch of inputs with character offsets.
777
778 This is an async version of encode_batch that can be awaited in async Python code.
779
780 Example:
781 Here are some examples of the inputs that are accepted::
782
783 await async_encode_batch([
784 "A single sequence",
785 ("A tuple with a sequence", "And its pair"),
786 [ "A", "pre", "tokenized", "sequence" ],
787 ([ "A", "pre", "tokenized", "sequence" ], "And its pair")
788 ])
789
790 Args:
791 input (A :obj:`List`/:obj:`Tuple` of :obj:`~tokenizers.EncodeInput`):
792 A list of single sequences or pair sequences to encode. Each sequence
793 can be either raw text or pre-tokenized, according to the ``is_pretokenized``
794 argument:
795
796 - If ``is_pretokenized=False``: :class:`~tokenizers.TextEncodeInput`
797 - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedEncodeInput`
798
799 is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
800 Whether the input is already pre-tokenized
801
802 add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
803 Whether to add the special tokens
804
805 Returns:
806 A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch
807 """
808 def async_encode_batch_fast(
809 self, /, input: Sequence[Any], is_pretokenized: bool = False, add_special_tokens: bool = True
810 ) -> Any:
811 """
812 Asynchronously encode the given batch of inputs without tracking character offsets.
813
814 This is an async version of encode_batch_fast that can be awaited in async Python code.
815
816 Example:
817 Here are some examples of the inputs that are accepted::
818
819 await async_encode_batch_fast([
820 "A single sequence",
821 ("A tuple with a sequence", "And its pair"),
822 [ "A", "pre", "tokenized", "sequence" ],
823 ([ "A", "pre", "tokenized", "sequence" ], "And its pair")
824 ])
825
826 Args:
827 input (A :obj:`List`/:obj:`Tuple` of :obj:`~tokenizers.EncodeInput`):
828 A list of single sequences or pair sequences to encode. Each sequence
829 can be either raw text or pre-tokenized, according to the ``is_pretokenized``
830 argument:
831
832 - If ``is_pretokenized=False``: :class:`~tokenizers.TextEncodeInput`
833 - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedEncodeInput`
834
835 is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
836 Whether the input is already pre-tokenized
837
838 add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
839 Whether to add the special tokens
840
841 Returns:
842 A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch
843 """
844 def decode(self, /, ids: Sequence[int], skip_special_tokens: bool = True) -> "str":
845 """
846 Decode the given list of ids back to a string
847
848 This is used to decode anything coming back from a Language Model
849
850 Args:
851 ids (A :obj:`List/Tuple` of :obj:`int`):
852 The list of ids that we want to decode
853
854 skip_special_tokens (:obj:`bool`, defaults to :obj:`True`):
855 Whether the special tokens should be removed from the decoded string
856
857 Returns:
858 :obj:`str`: The decoded string
859 """
860 def decode_batch(self, /, sequences: Sequence[Sequence[int]], skip_special_tokens: bool = True) -> "list[str]":
861 """
862 Decode a batch of ids back to their corresponding string
863
864 Args:
865 sequences (:obj:`List` of :obj:`List[int]`):
866 The batch of sequences we want to decode
867
868 skip_special_tokens (:obj:`bool`, defaults to :obj:`True`):
869 Whether the special tokens should be removed from the decoded strings
870
871 Returns:
872 :obj:`List[str]`: A list of decoded strings
873 """
874 @property
875 def decoder(self, /) -> Any:
876 """
877 The `optional` :class:`~tokenizers.decoders.Decoder` in use by the Tokenizer
878 """
879 @decoder.setter
880 def decoder(self, /, decoder: Decoder | None) -> None:
881 """
882 Set the :class:`~tokenizers.decoders.Decoder`
883 """
884 def enable_padding(self, /, **kwargs) -> "None":
885 """
886 Enable the padding
887
888 Args:
889 direction (:obj:`str`, `optional`, defaults to :obj:`right`):
890 The direction in which to pad. Can be either ``right`` or ``left``
891
892 pad_to_multiple_of (:obj:`int`, `optional`):
893 If specified, the padding length should always snap to the next multiple of the
894 given value. For example if we were going to pad witha length of 250 but
895 ``pad_to_multiple_of=8`` then we will pad to 256.
896
897 pad_id (:obj:`int`, defaults to 0):
898 The id to be used when padding
899
900 pad_type_id (:obj:`int`, defaults to 0):
901 The type id to be used when padding
902
903 pad_token (:obj:`str`, defaults to :obj:`[PAD]`):
904 The pad token to be used when padding
905
906 length (:obj:`int`, `optional`):
907 If specified, the length at which to pad. If not specified we pad using the size of
908 the longest sequence in a batch.
909 """
910 def enable_truncation(self, /, max_length: int, **kwargs) -> "None":
911 """
912 Enable truncation
913
914 Args:
915 max_length (:obj:`int`):
916 The max length at which to truncate
917
918 stride (:obj:`int`, `optional`):
919 The length of the previous first sequence to be included in the overflowing
920 sequence
921
922 strategy (:obj:`str`, `optional`, defaults to :obj:`longest_first`):
923 The strategy used to truncation. Can be one of ``longest_first``, ``only_first`` or
924 ``only_second``.
925
926 direction (:obj:`str`, defaults to :obj:`right`):
927 Truncate direction
928 """
929 def encode(
930 self, /, sequence: Any, pair: Any | None = None, is_pretokenized: bool = False, add_special_tokens: bool = True
931 ) -> "Encoding":
932 """
933 Encode the given sequence and pair. This method can process raw text sequences
934 as well as already pre-tokenized sequences.
935
936 Example:
937 Here are some examples of the inputs that are accepted::
938
939 encode("A single sequence")`
940 encode("A sequence", "And its pair")`
941 encode([ "A", "pre", "tokenized", "sequence" ], is_pretokenized=True)`
942 encode(
943 [ "A", "pre", "tokenized", "sequence" ], [ "And", "its", "pair" ],
944 is_pretokenized=True
945 )
946
947 Args:
948 sequence (:obj:`~tokenizers.InputSequence`):
949 The main input sequence we want to encode. This sequence can be either raw
950 text or pre-tokenized, according to the ``is_pretokenized`` argument:
951
952 - If ``is_pretokenized=False``: :class:`~tokenizers.TextInputSequence`
953 - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedInputSequence`
954
955 pair (:obj:`~tokenizers.InputSequence`, `optional`):
956 An optional input sequence. The expected format is the same that for ``sequence``.
957
958 is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
959 Whether the input is already pre-tokenized
960
961 add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
962 Whether to add the special tokens
963
964 Returns:
965 :class:`~tokenizers.Encoding`: The encoded result
966 """
967 def encode_batch(
968 self, /, input: Sequence[Any], is_pretokenized: bool = False, add_special_tokens: bool = True
969 ) -> "list[Encoding]":
970 """
971 Encode the given batch of inputs. This method accept both raw text sequences
972 as well as already pre-tokenized sequences. The reason we use `PySequence` is
973 because it allows type checking with zero-cost (according to PyO3) as we don't
974 have to convert to check.
975
976 Example:
977 Here are some examples of the inputs that are accepted::
978
979 encode_batch([
980 "A single sequence",
981 ("A tuple with a sequence", "And its pair"),
982 [ "A", "pre", "tokenized", "sequence" ],
983 ([ "A", "pre", "tokenized", "sequence" ], "And its pair")
984 ])
985
986 Args:
987 input (A :obj:`List`/:obj:`Tuple` of :obj:`~tokenizers.EncodeInput`):
988 A list of single sequences or pair sequences to encode. Each sequence
989 can be either raw text or pre-tokenized, according to the ``is_pretokenized``
990 argument:
991
992 - If ``is_pretokenized=False``: :class:`~tokenizers.TextEncodeInput`
993 - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedEncodeInput`
994
995 is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
996 Whether the input is already pre-tokenized
997
998 add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
999 Whether to add the special tokens
1000
1001 Returns:
1002 A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch
1003 """
1004 def encode_batch_fast(
1005 self, /, input: Sequence[Any], is_pretokenized: bool = False, add_special_tokens: bool = True
1006 ) -> "list[Encoding]":
1007 """
1008 Encode the given batch of inputs. This method is faster than `encode_batch`
1009 because it doesn't keep track of offsets, they will be all zeros.
1010
1011 Example:
1012 Here are some examples of the inputs that are accepted::
1013
1014 encode_batch_fast([
1015 "A single sequence",
1016 ("A tuple with a sequence", "And its pair"),
1017 [ "A", "pre", "tokenized", "sequence" ],
1018 ([ "A", "pre", "tokenized", "sequence" ], "And its pair")
1019 ])
1020
1021 Args:
1022 input (A :obj:`List`/:obj:`Tuple` of :obj:`~tokenizers.EncodeInput`):
1023 A list of single sequences or pair sequences to encode. Each sequence
1024 can be either raw text or pre-tokenized, according to the ``is_pretokenized``
1025 argument:
1026
1027 - If ``is_pretokenized=False``: :class:`~tokenizers.TextEncodeInput`
1028 - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedEncodeInput`
1029
1030 is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
1031 Whether the input is already pre-tokenized
1032
1033 add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
1034 Whether to add the special tokens
1035
1036 Returns:
1037 A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch
1038 """
1039 @property
1040 def encode_special_tokens(self, /) -> bool:
1041 """
1042 Get the value of the `encode_special_tokens` attribute
1043
1044 Returns:
1045 :obj:`bool`: the tokenizer's encode_special_tokens attribute
1046 """
1047 @encode_special_tokens.setter
1048 def encode_special_tokens(self, /, value: bool) -> None:
1049 """
1050 Modifies the tokenizer in order to use or not the special tokens
1051 during encoding.
1052
1053 Args:
1054 value (:obj:`bool`):
1055 Whether to use the special tokens or not
1056 """
1057 @staticmethod
1058 def from_buffer(buffer: bytes) -> "Tokenizer":
1059 """
1060 Instantiate a new :class:`~tokenizers.Tokenizer` from the given buffer.
1061
1062 Args:
1063 buffer (:obj:`bytes`):
1064 A buffer containing a previously serialized :class:`~tokenizers.Tokenizer`
1065
1066 Returns:
1067 :class:`~tokenizers.Tokenizer`: The new tokenizer
1068 """
1069 @staticmethod
1070 def from_file(path: str) -> "Tokenizer":
1071 """
1072 Instantiate a new :class:`~tokenizers.Tokenizer` from the file at the given path.
1073
1074 Args:
1075 path (:obj:`str`):
1076 A path to a local JSON file representing a previously serialized
1077 :class:`~tokenizers.Tokenizer`
1078
1079 Returns:
1080 :class:`~tokenizers.Tokenizer`: The new tokenizer
1081 """
1082 @staticmethod
1083 def from_pretrained(identifier: str, revision: str = ..., token: str | None = None) -> "Tokenizer":
1084 """
1085 Instantiate a new :class:`~tokenizers.Tokenizer` from an existing file on the
1086 Hugging Face Hub.
1087
1088 Args:
1089 identifier (:obj:`str`):
1090 The identifier of a Model on the Hugging Face Hub, that contains
1091 a tokenizer.json file
1092 revision (:obj:`str`, defaults to `main`):
1093 A branch or commit id
1094 token (:obj:`str`, `optional`, defaults to `None`):
1095 An optional auth token used to access private repositories on the
1096 Hugging Face Hub
1097
1098 Returns:
1099 :class:`~tokenizers.Tokenizer`: The new tokenizer
1100 """
1101 @staticmethod
1102 def from_str(json: str) -> "Tokenizer":
1103 """
1104 Instantiate a new :class:`~tokenizers.Tokenizer` from the given JSON string.
1105
1106 Args:
1107 json (:obj:`str`):
1108 A valid JSON string representing a previously serialized
1109 :class:`~tokenizers.Tokenizer`
1110
1111 Returns:
1112 :class:`~tokenizers.Tokenizer`: The new tokenizer
1113 """
1114 def get_added_tokens_decoder(self, /) -> "dict[int, AddedToken]":
1115 """
1116 Get the underlying vocabulary
1117
1118 Returns:
1119 :obj:`Dict[int, AddedToken]`: The vocabulary
1120 """
1121 def get_vocab(self, /, with_added_tokens: bool = True) -> "dict[str, int]":
1122 """
1123 Get the underlying vocabulary
1124
1125 Args:
1126 with_added_tokens (:obj:`bool`, defaults to :obj:`True`):
1127 Whether to include the added tokens
1128
1129 Returns:
1130 :obj:`Dict[str, int]`: The vocabulary
1131 """
1132 def get_vocab_size(self, /, with_added_tokens: bool = True) -> "int":
1133 """
1134 Get the size of the underlying vocabulary
1135
1136 Args:
1137 with_added_tokens (:obj:`bool`, defaults to :obj:`True`):
1138 Whether to include the added tokens
1139
1140 Returns:
1141 :obj:`int`: The size of the vocabulary
1142 """
1143 def id_to_token(self, /, id: int) -> "str | None":
1144 """
1145 Convert the given id to its corresponding token if it exists
1146
1147 Args:
1148 id (:obj:`int`):
1149 The id to convert
1150
1151 Returns:
1152 :obj:`Optional[str]`: An optional token, :obj:`None` if out of vocabulary
1153 """
1154 @property
1155 def model(self, /) -> Any:
1156 """
1157 The :class:`~tokenizers.models.Model` in use by the Tokenizer
1158 """
1159 @model.setter
1160 def model(self, /, model: Model) -> None:
1161 """
1162 Set the :class:`~tokenizers.models.Model`
1163 """
1164 def no_padding(self, /) -> None:
1165 """
1166 Disable padding
1167 """
1168 def no_truncation(self, /) -> None:
1169 """
1170 Disable truncation
1171 """
1172 @property
1173 def normalizer(self, /) -> Any:
1174 """
1175 The `optional` :class:`~tokenizers.normalizers.Normalizer` in use by the Tokenizer
1176 """
1177 @normalizer.setter
1178 def normalizer(self, /, normalizer: Normalizer | None) -> None:
1179 """
1180 Set the :class:`~tokenizers.normalizers.Normalizer`
1181 """
1182 def num_special_tokens_to_add(self, /, is_pair: bool) -> int:
1183 """
1184 Return the number of special tokens that would be added for single/pair sentences.
1185 :param is_pair: Boolean indicating if the input would be a single sentence or a pair
1186 :return:
1187 """
1188 @property
1189 def padding(self, /) -> dict | None:
1190 """
1191 Get the current padding parameters
1192
1193 `Cannot be set, use` :meth:`~tokenizers.Tokenizer.enable_padding` `instead`
1194
1195 Returns:
1196 (:obj:`dict`, `optional`):
1197 A dict with the current padding parameters if padding is enabled
1198 """
1199 def post_process(
1200 self, /, encoding: Encoding, pair: Encoding | None = None, add_special_tokens: bool = True
