Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
__init__.pyi1329 linesDownload Raw Back to tokenizers
1"""
2Tokenizers Module
3"""
4
5from _typeshed import Incomplete
6from collections.abc import Sequence
7from tokenizers.decoders import Decoder
8from tokenizers.models import Model
9from tokenizers.normalizers import Normalizer
10from tokenizers.pre_tokenizers import PreTokenizer
11from tokenizers.processors import PostProcessor
12from tokenizers.trainers import Trainer
13from typing import Any, Final, final
14
15__version__: Final[str]
16
17@final
18class AddedToken:
19    """
20    Represents a token that can be be added to a :class:`~tokenizers.Tokenizer`.
21    It can have special options that defines the way it should behave.
22
23    Args:
24        content (:obj:`str`): The content of the token
25
26        single_word (:obj:`bool`, defaults to :obj:`False`):
27            Defines whether this token should only match single words. If :obj:`True`, this
28            token will never match inside of a word. For example the token ``ing`` would match
29            on ``tokenizing`` if this option is :obj:`False`, but not if it is :obj:`True`.
30            The notion of "`inside of a word`" is defined by the word boundaries pattern in
31            regular expressions (ie. the token should start and end with word boundaries).
32
33        lstrip (:obj:`bool`, defaults to :obj:`False`):
34            Defines whether this token should strip all potential whitespaces on its left side.
35            If :obj:`True`, this token will greedily match any whitespace on its left. For
36            example if we try to match the token ``[MASK]`` with ``lstrip=True``, in the text
37            ``"I saw a [MASK]"``, we would match on ``" [MASK]"``. (Note the space on the left).
38
39        rstrip (:obj:`bool`, defaults to :obj:`False`):
40            Defines whether this token should strip all potential whitespaces on its right
41            side. If :obj:`True`, this token will greedily match any whitespace on its right.
42            It works just like :obj:`lstrip` but on the right.
43
44        normalized (:obj:`bool`, defaults to :obj:`True` with :meth:`~tokenizers.Tokenizer.add_tokens` and :obj:`False` with :meth:`~tokenizers.Tokenizer.add_special_tokens`):
45            Defines whether this token should match against the normalized version of the input
46            text. For example, with the added token ``"yesterday"``, and a normalizer in charge of
47            lowercasing the text, the token could be extract from the input ``"I saw a lion
48            Yesterday"``.
49        special (:obj:`bool`, defaults to :obj:`False` with :meth:`~tokenizers.Tokenizer.add_tokens` and :obj:`False` with :meth:`~tokenizers.Tokenizer.add_special_tokens`):
50            Defines whether this token should be skipped when decoding.
51    """
52    def __eq__(self, /, other: object) -> bool: ...
53    def __ge__(self, /, other: object) -> bool: ...
54    def __getstate__(self, /) -> dict: ...
55    def __gt__(self, /, other: object) -> bool: ...
56    def __hash__(self, /) -> int: ...
57    def __le__(self, /, other: object) -> bool: ...
58    def __lt__(self, /, other: object) -> bool: ...
59    def __ne__(self, /, other: object) -> bool: ...
60    def __new__(cls, /, content: str | None = None, **kwargs) -> AddedToken: ...
61    def __repr__(self, /) -> str: ...
62    def __setstate__(self, /, state: Any) -> None: ...
63    def __str__(self, /) -> str: ...
64    @property
65    def content(self, /) -> str:
66        """
67        Get the content of this :obj:`AddedToken`
68        """
69    @content.setter
70    def content(self, /, content: str) -> None:
71        """
72        Set the content of this :obj:`AddedToken`
73        """
74    @property
75    def lstrip(self, /) -> bool:
76        """
77        Get the value of the :obj:`lstrip` option
78        """
79    @property
80    def normalized(self, /) -> bool:
81        """
82        Get the value of the :obj:`normalized` option
83        """
84    @property
85    def rstrip(self, /) -> bool:
86        """
87        Get the value of the :obj:`rstrip` option
88        """
89    @property
90    def single_word(self, /) -> bool:
91        """
92        Get the value of the :obj:`single_word` option
93        """
94    @property
95    def special(self, /) -> bool:
96        """
97        Get the value of the :obj:`special` option
98        """
99    @special.setter
100    def special(self, /, special: bool) -> None:
101        """
102        Set the value of the :obj:`special` option
103        """
104
105@final
106class Encoding:
107    """
108    The :class:`~tokenizers.Encoding` represents the output of a :class:`~tokenizers.Tokenizer`.
109
110    It holds all the information about the tokenized input, including the token IDs,
111    token strings, attention masks, offsets, and more. This is the main data structure
112    returned by :meth:`~tokenizers.Tokenizer.encode` and
113    :meth:`~tokenizers.Tokenizer.encode_batch`.
114
115    Example::
116
117        >>> from tokenizers import Tokenizer
118        >>> tokenizer = Tokenizer.from_pretrained("bert-base-uncased")
119        >>> encoding = tokenizer.encode("Hello, world!")
120        >>> encoding.ids
121        [101, 7592, 1010, 2088, 999, 102]
122        >>> encoding.tokens
123        ['[CLS]', 'hello', ',', 'world', '!', '[SEP]']
124        >>> encoding.offsets
125        [(0, 0), (0, 5), (5, 6), (7, 12), (12, 13), (0, 0)]
126    """
127    def __getstate__(self, /) -> Any: ...
128    def __len__(self, /) -> int: ...
129    def __new__(cls, /) -> Encoding: ...
130    def __repr__(self, /) -> str: ...
131    def __setstate__(self, /, state: Any) -> None: ...
132    @property
133    def attention_mask(self, /) -> list[int]:
134        """
135        The attention mask
136
137        This indicates to the LM which tokens should be attended to, and which should not.
138        This is especially important when batching sequences, where we need to applying
139        padding.
140
141        Returns:
142           :obj:`List[int]`: The attention mask
143        """
144    def char_to_token(self, /, char_pos: int, sequence_index: int = 0) -> int | None:
145        """
146        Get the token that contains the char at the given position in the input sequence.
147
148        Args:
149            char_pos (:obj:`int`):
150                The position of a char in the input string
151            sequence_index (:obj:`int`, defaults to :obj:`0`):
152                The index of the sequence that contains the target char
153
154        Returns:
155            :obj:`int`: The index of the token that contains this char in the encoded sequence
156        """
157    def char_to_word(self, /, char_pos: int, sequence_index: int = 0) -> int | None:
158        """
159        Get the word that contains the char at the given position in the input sequence.
160
161        Args:
162            char_pos (:obj:`int`):
163                The position of a char in the input string
164            sequence_index (:obj:`int`, defaults to :obj:`0`):
165                The index of the sequence that contains the target char
166
167        Returns:
168            :obj:`int`: The index of the word that contains this char in the input sequence
169        """
170    @property
171    def ids(self, /) -> list[int]:
172        """
173        The generated IDs
174
175        The IDs are the main input to a Language Model. They are the token indices,
176        the numerical representations that a LM understands.
177
178        Returns:
179            :obj:`List[int]`: The list of IDs
180        """
181    @staticmethod
182    def merge(encodings: Sequence[Encoding], growing_offsets: bool = True) -> "Encoding":
183        """
184        Merge the list of encodings into one final :class:`~tokenizers.Encoding`
185
186        Args:
187            encodings (A :obj:`List` of :class:`~tokenizers.Encoding`):
188                The list of encodings that should be merged in one
189
190            growing_offsets (:obj:`bool`, defaults to :obj:`True`):
191                Whether the offsets should accumulate while merging
192
193        Returns:
194            :class:`~tokenizers.Encoding`: The resulting Encoding
195        """
196    @property
197    def n_sequences(self, /) -> int:
198        """
199        The number of sequences represented
200
201        Returns:
202            :obj:`int`: The number of sequences in this :class:`~tokenizers.Encoding`
203        """
204    @property
205    def offsets(self, /) -> list[tuple[int, int]]:
206        """
207        The offsets associated to each token
208
209        These offsets let's you slice the input string, and thus retrieve the original
210        part that led to producing the corresponding token.
211
212        Returns:
213            A :obj:`List` of :obj:`Tuple[int, int]`: The list of offsets
214        """
215    @property
216    def overflowing(self, /) -> list[Encoding]:
217        """
218        A :obj:`List` of overflowing :class:`~tokenizers.Encoding`
219
220        When using truncation, the :class:`~tokenizers.Tokenizer` takes care of splitting
221        the output into as many pieces as required to match the specified maximum length.
222        This field lets you retrieve all the subsequent pieces.
223
224        When you use pairs of sequences, the overflowing pieces will contain enough
225        variations to cover all the possible combinations, while respecting the provided
226        maximum length.
227        """
228    def pad(self, /, length: int, **kwargs) -> "None":
229        """
230        Pad the :class:`~tokenizers.Encoding` at the given length
231
232        Args:
233            length (:obj:`int`):
234                The desired length
235
236            direction: (:obj:`str`, defaults to :obj:`right`):
237                The expected padding direction. Can be either :obj:`right` or :obj:`left`
238
239            pad_id (:obj:`int`, defaults to :obj:`0`):
240                The ID corresponding to the padding token
241
242            pad_type_id (:obj:`int`, defaults to :obj:`0`):
243                The type ID corresponding to the padding token
244
245            pad_token (:obj:`str`, defaults to `[PAD]`):
246                The pad token to use
247        """
248    @property
249    def sequence_ids(self, /) -> list[int | None]:
250        """
251        The generated sequence indices.
252
253        They represent the index of the input sequence associated to each token.
254        The sequence id can be None if the token is not related to any input sequence,
255        like for example with special tokens.
256
257        Returns:
258            A :obj:`List` of :obj:`Optional[int]`: A list of optional sequence index.
259        """
260    def set_sequence_id(self, /, sequence_id: int) -> None:
261        """
262        Set the given sequence index
263
264        Set the given sequence index for the whole range of tokens contained in this
265        :class:`~tokenizers.Encoding`.
266        """
267    @property
268    def special_tokens_mask(self, /) -> list[int]:
269        """
270        The special token mask
271
272        This indicates which tokens are special tokens, and which are not.
273
274        Returns:
275            :obj:`List[int]`: The special tokens mask
276        """
277    def token_to_chars(self, /, token_index: int) -> tuple[int, int] | None:
278        """
279        Get the offsets of the token at the given index.
280
281        The returned offsets are related to the input sequence that contains the
282        token.  In order to determine in which input sequence it belongs, you
283        must call :meth:`~tokenizers.Encoding.token_to_sequence()`.
284
285        Args:
286            token_index (:obj:`int`):
287                The index of a token in the encoded sequence.
288
289        Returns:
290            :obj:`Tuple[int, int]`: The token offsets :obj:`(first, last + 1)`
291        """
292    def token_to_sequence(self, /, token_index: int) -> int | None:
293        """
294        Get the index of the sequence represented by the given token.
295
296        In the general use case, this method returns :obj:`0` for a single sequence or
297        the first sequence of a pair, and :obj:`1` for the second sequence of a pair
298
299        Args:
300            token_index (:obj:`int`):
301                The index of a token in the encoded sequence.
302
303        Returns:
304            :obj:`int`: The sequence id of the given token
305        """
306    def token_to_word(self, /, token_index: int) -> int | None:
307        """
308        Get the index of the word that contains the token in one of the input sequences.
309
310        The returned word index is related to the input sequence that contains
311        the token.  In order to determine in which input sequence it belongs, you
312        must call :meth:`~tokenizers.Encoding.token_to_sequence()`.
313
314        Args:
315            token_index (:obj:`int`):
316                The index of a token in the encoded sequence.
317
318        Returns:
319            :obj:`int`: The index of the word in the relevant input sequence.
320        """
321    @property
322    def tokens(self, /) -> list[str]:
323        """
324        The generated tokens
325
326        They are the string representation of the IDs.
327
328        Returns:
329            :obj:`List[str]`: The list of tokens
330        """
331    def truncate(self, /, max_length: int, stride: int = 0, direction: str = "right") -> "None":
332        """
333        Truncate the :class:`~tokenizers.Encoding` at the given length
334
335        If this :class:`~tokenizers.Encoding` represents multiple sequences, when truncating
336        this information is lost. It will be considered as representing a single sequence.
337
338        Args:
339            max_length (:obj:`int`):
340                The desired length
341
342            stride (:obj:`int`, defaults to :obj:`0`):
343                The length of previous content to be included in each overflowing piece
344
345            direction (:obj:`str`, defaults to :obj:`right`):
346                Truncate direction
347        """
348    @property
349    def type_ids(self, /) -> list[int]:
350        """
351        The generated type IDs
352
353        Generally used for tasks like sequence classification or question answering,
354        these tokens let the LM know which input sequence corresponds to each tokens.
355
356        Returns:
357            :obj:`List[int]`: The list of type ids
358        """
359    @property
360    def word_ids(self, /) -> list[int | None]:
361        """
362        The generated word indices.
363
364        They represent the index of the word associated to each token.
365        When the input is pre-tokenized, they correspond to the ID of the given input label,
366        otherwise they correspond to the words indices as defined by the
367        :class:`~tokenizers.pre_tokenizers.PreTokenizer` that was used.
368
369        For special tokens and such (any token that was generated from something that was
370        not part of the input), the output is :obj:`None`
371
372        Returns:
373            A :obj:`List` of :obj:`Optional[int]`: A list of optional word index.
374        """
375    def word_to_chars(self, /, word_index: int, sequence_index: int = 0) -> tuple[int, int] | None:
376        """
377        Get the offsets of the word at the given index in one of the input sequences.
378
379        Args:
380            word_index (:obj:`int`):
381                The index of a word in one of the input sequences.
382            sequence_index (:obj:`int`, defaults to :obj:`0`):
383                The index of the sequence that contains the target word
384
385        Returns:
386            :obj:`Tuple[int, int]`: The range of characters (span) :obj:`(first, last + 1)`
387        """
388    def word_to_tokens(self, /, word_index: int, sequence_index: int = 0) -> tuple[int, int] | None:
389        """
390        Get the encoded tokens corresponding to the word at the given index
391        in one of the input sequences.
392
393        Args:
394            word_index (:obj:`int`):
395                The index of a word in one of the input sequences.
396            sequence_index (:obj:`int`, defaults to :obj:`0`):
397                The index of the sequence that contains the target word
398
399        Returns:
400            :obj:`Tuple[int, int]`: The range of tokens: :obj:`(first, last + 1)`
401        """
402    @property
403    def words(self, /) -> list[int | None]:
404        """
405        The generated word indices.
406
407        .. warning::
408            This is deprecated and will be removed in a future version.
409            Please use :obj:`~tokenizers.Encoding.word_ids` instead.
410
411        They represent the index of the word associated to each token.
412        When the input is pre-tokenized, they correspond to the ID of the given input label,
413        otherwise they correspond to the words indices as defined by the
414        :class:`~tokenizers.pre_tokenizers.PreTokenizer` that was used.
415
416        For special tokens and such (any token that was generated from something that was
417        not part of the input), the output is :obj:`None`
418
419        Returns:
420            A :obj:`List` of :obj:`Optional[int]`: A list of optional word index.
421        """
422
423@final
424class NormalizedString:
425    """
426    NormalizedString
427
428    A NormalizedString takes care of modifying an "original" string, to obtain a "normalized" one.
429    While making all the requested modifications, it keeps track of the alignment information
430    between the two versions of the string.
431
432    Args:
433        sequence: str:
434            The string sequence used to initialize this NormalizedString
435    """
436    def __getitem__(self, /, range: int | tuple[int, int] | slice) -> NormalizedString | None: ...
437    def __new__(cls, /, sequence: str) -> NormalizedString: ...
438    def __repr__(self, /) -> str: ...
439    def __str__(self, /) -> str: ...
440    def append(self, /, s: str) -> None:
441        """
442        Append the given sequence to the string
443        """
444    def clear(self, /) -> None:
445        """
446        Clears the string
447        """
448    def filter(self, /, func: Any) -> None:
449        """
450        Filter each character of the string using the given func
451        """
452    def for_each(self, /, func: Any) -> None:
453        """
454        Calls the given function for each character of the string
455        """
456    def lowercase(self, /) -> None:
457        """
458        Lowercase the string
459        """
460    def lstrip(self, /) -> None:
461        """
462        Strip the left of the string
463        """
464    def map(self, /, func: Any) -> None:
465        """
466        Calls the given function for each character of the string
467
468        Replaces each character of the string using the returned value. Each
469        returned value **must** be a str of length 1 (ie a character).
470        """
471    def nfc(self, /) -> None:
472        """
473        Runs the NFC normalization
474        """
475    def nfd(self, /) -> None:
476        """
477        Runs the NFD normalization
478        """
479    def nfkc(self, /) -> None:
480        """
481        Runs the NFKC normalization
482        """
483    def nfkd(self, /) -> None:
484        """
485        Runs the NFKD normalization
486        """
487    @property
488    def normalized(self, /) -> str:
489        """
490        The normalized part of the string
491        """
492    @property
493    def original(self, /) -> str: ...
494    def prepend(self, /, s: str) -> None:
495        """
496        Prepend the given sequence to the string
497        """
498    def replace(self, /, pattern: str | Regex, content: str) -> None:
499        """
500        Replace the content of the given pattern with the provided content
501
502        Args:
503            pattern: Pattern:
504                A pattern used to match the string. Usually a string or a Regex
505
506            content: str:
507                The content to be used as replacement
508        """
509    def rstrip(self, /) -> None:
510        """
511        Strip the right of the string
512        """
513    def slice(self, /, range: int | tuple[int, int] | slice) -> NormalizedString | None:
514        """
515        Slice the string using the given range
516        """
517    def split(self, /, pattern: str | Regex, behavior: Incomplete) -> list[NormalizedString]:
518        """
519        Split the NormalizedString using the given pattern and the specified behavior
520
521        Args:
522            pattern: Pattern:
523                A pattern used to split the string. Usually a string or a regex built with `tokenizers.Regex`
524
525            behavior: SplitDelimiterBehavior:
526                The behavior to use when splitting.
527                Choices: "removed", "isolated", "merged_with_previous", "merged_with_next",
528                "contiguous"
529
530        Returns:
531            A list of NormalizedString, representing each split
532        """
533    def strip(self, /) -> None:
534        """
535        Strip both ends of the string
536        """
537    def uppercase(self, /) -> None:
538        """
539        Uppercase the string
540        """
541
542@final
543class PreTokenizedString:
544    """
545    PreTokenizedString
546
547    Wrapper over a string, that provides a way to normalize, pre-tokenize, tokenize the
548    underlying string, while keeping track of the alignment information (offsets).
549
550    The PreTokenizedString manages what we call `splits`. Each split represents a substring
551    which is a subpart of the original string, with the relevant offsets and tokens.
552
553    When calling one of the methods used to modify the PreTokenizedString (namely one of
554    `split`, `normalize` or `tokenize), only the `splits` that don't have any associated
555    tokens will get modified.
556
557    Args:
558        sequence: str:
559            The string sequence used to initialize this PreTokenizedString
560    """
561    def __new__(cls, /, s: str) -> PreTokenizedString: ...
562    def get_splits(
563        self, /, offset_referential: Incomplete = ..., offset_type: Incomplete = ...
564    ) -> list[tuple[str, tuple[int, int], list[Token] | None]]:
565        """
566        Get the splits currently managed by the PreTokenizedString
567
568        Args:
569            offset_referential: :obj:`str`
570                Whether the returned splits should have offsets expressed relative
571                to the original string, or the normalized one. choices: "original", "normalized".
572
573            offset_type: :obj:`str`
574                Whether the returned splits should have offsets expressed in bytes or chars.
575                When slicing an str, we usually want to use chars, which is the default value.
576                Now in some cases it might be interesting to get these offsets expressed in bytes,
577                so it is possible to change this here.
578                choices: "char", "bytes"
579
580        Returns
581            A list of splits
582        """
583    def normalize(self, /, func: Any) -> None:
584        """
585        Normalize each split of the `PreTokenizedString` using the given `func`
586
587        Args:
588            func: Callable[[NormalizedString], None]:
589                The function used to normalize each underlying split. This function
590                does not need to return anything, just calling the methods on the provided
591                NormalizedString allow its modification.
592        """
593    def split(self, /, func: Any) -> None:
594        """
595        Split the PreTokenizedString using the given `func`
596
597        Args:
598            func: Callable[[index, NormalizedString], List[NormalizedString]]:
599                The function used to split each underlying split.
600                It is expected to return a list of `NormalizedString`, that represent the new
601                splits. If the given `NormalizedString` does not need any splitting, we can
602                just return it directly.
603                In order for the offsets to be tracked accurately, any returned `NormalizedString`
604                should come from calling either `.split` or `.slice` on the received one.
605        """
606    def to_encoding(self, /, type_id: int = 0, word_idx: int | None = None) -> "Encoding":
607        """
608        Return an Encoding generated from this PreTokenizedString
609
610        Args:
611            type_id: int = 0:
612                The type_id to be used on the generated Encoding.
613
614            word_idx: Optional[int] = None:
615                An optional word index to be used for each token of this Encoding. If provided,
616                all the word indices in the generated Encoding will use this value, instead
617                of the one automatically tracked during pre-tokenization.
618
619        Returns:
620            An Encoding
621        """
622    def tokenize(self, /, func: Any) -> None:
623        """
624        Tokenize each split of the `PreTokenizedString` using the given `func`
625
626        Args:
627            func: Callable[[str], List[Token]]:
628                The function used to tokenize each underlying split. This function must return
629                a list of Token generated from the input str.
630        """
631
632@final
633class Regex:
634    """
635    Instantiate a new Regex with the given pattern
636    """
637    def __new__(cls, /, s: str) -> Regex: ...
638
639@final
640class Token:
641    def __new__(cls, /, id: int, value: str, offsets: tuple[int, int]) -> Token:
642        """
643        Create a token from id, string value and byte offsets
644        """
645    def as_tuple(self, /) -> tuple[int, str, tuple[int, int]]: ...
646    @property
647    def id(self, /) -> int: ...
648    @property
649    def offsets(self, /) -> tuple[int, int]: ...
650    @property
651    def value(self, /) -> str: ...
652
653@final
654class Tokenizer:
655    """
656    A :obj:`Tokenizer` works as a pipeline. It processes some raw text as input
657    and outputs an :class:`~tokenizers.Encoding`.
658
659    The pipeline is structured as follows:
660
661        1. The :class:`~tokenizers.normalizers.Normalizer` normalizes the raw input text.
662        2. The :class:`~tokenizers.pre_tokenizers.PreTokenizer` splits the normalized text
663           into word-level tokens.
664        3. The :class:`~tokenizers.models.Model` tokenizes each word into subword tokens
665           and maps them to IDs.
666        4. The :class:`~tokenizers.processors.PostProcessor` applies any final
667           transformations (e.g., adding special tokens like ``[CLS]`` and ``[SEP]``).
668
669    Args:
670        model (:class:`~tokenizers.models.Model`):
671            The core algorithm that this :obj:`Tokenizer` should be using.
672
673    Example::
674
675        >>> from tokenizers import Tokenizer
676        >>> from tokenizers.models import BPE
677        >>> from tokenizers.normalizers import Lowercase
678        >>> from tokenizers.pre_tokenizers import Whitespace
679        >>> tokenizer = Tokenizer(BPE(unk_token="<unk>"))
680        >>> tokenizer.normalizer = Lowercase()
681        >>> tokenizer.pre_tokenizer = Whitespace()
682        >>> # Load a pre-built tokenizer from HuggingFace Hub
683        >>> tokenizer = Tokenizer.from_pretrained("bert-base-uncased")
684    """
685    def __getnewargs__(self, /) -> tuple: ...
686    def __getstate__(self, /) -> Any: ...
687    def __new__(cls, /, model: Model) -> Tokenizer: ...
688    def __repr__(self, /) -> str: ...
689    def __setstate__(self, /, state: Any) -> None: ...
690    def __str__(self, /) -> str: ...
691    def add_special_tokens(self, /, tokens: list) -> int:
692        """
693        Add the given special tokens to the Tokenizer.
694
695        If these tokens are already part of the vocabulary, it just let the Tokenizer know about
696        them. If they don't exist, the Tokenizer creates them, giving them a new id.
697
698        These special tokens will never be processed by the model (ie won't be split into
699        multiple tokens), and they can be removed from the output when decoding.
700
701        Args:
702            tokens (A :obj:`List` of :class:`~tokenizers.AddedToken` or :obj:`str`):
703                The list of special tokens we want to add to the vocabulary. Each token can either
704                be a string or an instance of :class:`~tokenizers.AddedToken` for more
705                customization.
706
707        Returns:
708            :obj:`int`: The number of tokens that were created in the vocabulary
709        """
710    def add_tokens(self, /, tokens: list) -> int:
711        """
712        Add the given tokens to the vocabulary
713
714        The given tokens are added only if they don't already exist in the vocabulary.
715        Each token then gets a new attributed id.
716
717        Args:
718            tokens (A :obj:`List` of :class:`~tokenizers.AddedToken` or :obj:`str`):
719                The list of tokens we want to add to the vocabulary. Each token can be either a
720                string or an instance of :class:`~tokenizers.AddedToken` for more customization.
721
722        Returns:
723            :obj:`int`: The number of tokens that were created in the vocabulary
724        """
725    def async_decode_batch(self, /, sequences: Sequence[Sequence[int]], skip_special_tokens: bool = True) -> Any:
726        """
727        Decode a batch of ids back to their corresponding string
728
729        Args:
730            sequences (:obj:`List` of :obj:`List[int]`):
731                The batch of sequences we want to decode
732
733            skip_special_tokens (:obj:`bool`, defaults to :obj:`True`):
734                Whether the special tokens should be removed from the decoded strings
735
736        Returns:
737            :obj:`List[str]`: A list of decoded strings
738        """
739    def async_encode(
740        self, /, sequence: Any, pair: Any | None = None, is_pretokenized: bool = False, add_special_tokens: bool = True
741    ) -> Any:
742        """
743        Asynchronously encode the given input with character offsets.
744
745        This is an async version of encode that can be awaited in async Python code.
746
747        Example:
748            Here are some examples of the inputs that are accepted::
749
750                await async_encode("A single sequence")
751
752        Args:
753            sequence (:obj:`~tokenizers.InputSequence`):
754                The main input sequence we want to encode. This sequence can be either raw
755                text or pre-tokenized, according to the ``is_pretokenized`` argument:
756
757                - If ``is_pretokenized=False``: :class:`~tokenizers.TextInputSequence`
758                - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedInputSequence`
759
760            pair (:obj:`~tokenizers.InputSequence`, `optional`):
761                An optional input sequence. The expected format is the same that for ``sequence``.
762
763            is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
764                Whether the input is already pre-tokenized
765
766            add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
767                Whether to add the special tokens
768
769        Returns:
770            :class:`~tokenizers.Encoding`: The encoded result
771        """
772    def async_encode_batch(
773        self, /, input: Sequence[Any], is_pretokenized: bool = False, add_special_tokens: bool = True
774    ) -> Any:
775        """
776        Asynchronously encode the given batch of inputs with character offsets.
777
778        This is an async version of encode_batch that can be awaited in async Python code.
779
780        Example:
781            Here are some examples of the inputs that are accepted::
782
783                await async_encode_batch([
784                    "A single sequence",
785                    ("A tuple with a sequence", "And its pair"),
786                    [ "A", "pre", "tokenized", "sequence" ],
787                    ([ "A", "pre", "tokenized", "sequence" ], "And its pair")
788                ])
789
790        Args:
791            input (A :obj:`List`/:obj:`Tuple` of :obj:`~tokenizers.EncodeInput`):
792                A list of single sequences or pair sequences to encode. Each sequence
793                can be either raw text or pre-tokenized, according to the ``is_pretokenized``
794                argument:
795
796                - If ``is_pretokenized=False``: :class:`~tokenizers.TextEncodeInput`
797                - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedEncodeInput`
798
799            is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
800                Whether the input is already pre-tokenized
801
802            add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
803                Whether to add the special tokens
804
805        Returns:
806            A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch
807        """
808    def async_encode_batch_fast(
809        self, /, input: Sequence[Any], is_pretokenized: bool = False, add_special_tokens: bool = True
810    ) -> Any:
811        """
812        Asynchronously encode the given batch of inputs without tracking character offsets.
813
814        This is an async version of encode_batch_fast that can be awaited in async Python code.
815
816        Example:
817            Here are some examples of the inputs that are accepted::
818
819                await async_encode_batch_fast([
820                    "A single sequence",
821                    ("A tuple with a sequence", "And its pair"),
822                    [ "A", "pre", "tokenized", "sequence" ],
823                    ([ "A", "pre", "tokenized", "sequence" ], "And its pair")
824                ])
825
826        Args:
827            input (A :obj:`List`/:obj:`Tuple` of :obj:`~tokenizers.EncodeInput`):
828                A list of single sequences or pair sequences to encode. Each sequence
829                can be either raw text or pre-tokenized, according to the ``is_pretokenized``
830                argument:
831
832                - If ``is_pretokenized=False``: :class:`~tokenizers.TextEncodeInput`
833                - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedEncodeInput`
834
835            is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
836                Whether the input is already pre-tokenized
837
838            add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
839                Whether to add the special tokens
840
841        Returns:
842            A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch
843        """
844    def decode(self, /, ids: Sequence[int], skip_special_tokens: bool = True) -> "str":
845        """
846        Decode the given list of ids back to a string
847
848        This is used to decode anything coming back from a Language Model
849
850        Args:
851            ids (A :obj:`List/Tuple` of :obj:`int`):
852                The list of ids that we want to decode
853
854            skip_special_tokens (:obj:`bool`, defaults to :obj:`True`):
855                Whether the special tokens should be removed from the decoded string
856
857        Returns:
858            :obj:`str`: The decoded string
859        """
860    def decode_batch(self, /, sequences: Sequence[Sequence[int]], skip_special_tokens: bool = True) -> "list[str]":
861        """
862        Decode a batch of ids back to their corresponding string
863
864        Args:
865            sequences (:obj:`List` of :obj:`List[int]`):
866                The batch of sequences we want to decode
867
868            skip_special_tokens (:obj:`bool`, defaults to :obj:`True`):
869                Whether the special tokens should be removed from the decoded strings
870
871        Returns:
872            :obj:`List[str]`: A list of decoded strings
873        """
874    @property
875    def decoder(self, /) -> Any:
876        """
877        The `optional` :class:`~tokenizers.decoders.Decoder` in use by the Tokenizer
878        """
879    @decoder.setter
880    def decoder(self, /, decoder: Decoder | None) -> None:
881        """
882        Set the :class:`~tokenizers.decoders.Decoder`
883        """
884    def enable_padding(self, /, **kwargs) -> "None":
885        """
886        Enable the padding
887
888        Args:
889            direction (:obj:`str`, `optional`, defaults to :obj:`right`):
890                The direction in which to pad. Can be either ``right`` or ``left``
891
892            pad_to_multiple_of (:obj:`int`, `optional`):
893                If specified, the padding length should always snap to the next multiple of the
894                given value. For example if we were going to pad witha length of 250 but
895                ``pad_to_multiple_of=8`` then we will pad to 256.
896
897            pad_id (:obj:`int`, defaults to 0):
898                The id to be used when padding
899
900            pad_type_id (:obj:`int`, defaults to 0):
901                The type id to be used when padding
902
903            pad_token (:obj:`str`, defaults to :obj:`[PAD]`):
904                The pad token to be used when padding
905
906            length (:obj:`int`, `optional`):
907                If specified, the length at which to pad. If not specified we pad using the size of
908                the longest sequence in a batch.
909        """
910    def enable_truncation(self, /, max_length: int, **kwargs) -> "None":
911        """
912        Enable truncation
913
914        Args:
915            max_length (:obj:`int`):
916                The max length at which to truncate
917
918            stride (:obj:`int`, `optional`):
919                The length of the previous first sequence to be included in the overflowing
920                sequence
921
922            strategy (:obj:`str`, `optional`, defaults to :obj:`longest_first`):
923                The strategy used to truncation. Can be one of ``longest_first``, ``only_first`` or
924                ``only_second``.
925
926            direction (:obj:`str`, defaults to :obj:`right`):
927                Truncate direction
928        """
929    def encode(
930        self, /, sequence: Any, pair: Any | None = None, is_pretokenized: bool = False, add_special_tokens: bool = True
931    ) -> "Encoding":
932        """
933        Encode the given sequence and pair. This method can process raw text sequences
934        as well as already pre-tokenized sequences.
935
936        Example:
937            Here are some examples of the inputs that are accepted::
938
939                encode("A single sequence")`
940                encode("A sequence", "And its pair")`
941                encode([ "A", "pre", "tokenized", "sequence" ], is_pretokenized=True)`
942                encode(
943                    [ "A", "pre", "tokenized", "sequence" ], [ "And", "its", "pair" ],
944                    is_pretokenized=True
945                )
946
947        Args:
948            sequence (:obj:`~tokenizers.InputSequence`):
949                The main input sequence we want to encode. This sequence can be either raw
950                text or pre-tokenized, according to the ``is_pretokenized`` argument:
951
952                - If ``is_pretokenized=False``: :class:`~tokenizers.TextInputSequence`
953                - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedInputSequence`
954
955            pair (:obj:`~tokenizers.InputSequence`, `optional`):
956                An optional input sequence. The expected format is the same that for ``sequence``.
957
958            is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
959                Whether the input is already pre-tokenized
960
961            add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
962                Whether to add the special tokens
963
964        Returns:
965            :class:`~tokenizers.Encoding`: The encoded result
966        """
967    def encode_batch(
968        self, /, input: Sequence[Any], is_pretokenized: bool = False, add_special_tokens: bool = True
969    ) -> "list[Encoding]":
970        """
971        Encode the given batch of inputs. This method accept both raw text sequences
972        as well as already pre-tokenized sequences. The reason we use `PySequence` is
973        because it allows type checking with zero-cost (according to PyO3) as we don't
974        have to convert to check.
975
976        Example:
977            Here are some examples of the inputs that are accepted::
978
979                encode_batch([
980                    "A single sequence",
981                    ("A tuple with a sequence", "And its pair"),
982                    [ "A", "pre", "tokenized", "sequence" ],
983                    ([ "A", "pre", "tokenized", "sequence" ], "And its pair")
984                ])
985
986        Args:
987            input (A :obj:`List`/:obj:`Tuple` of :obj:`~tokenizers.EncodeInput`):
988                A list of single sequences or pair sequences to encode. Each sequence
989                can be either raw text or pre-tokenized, according to the ``is_pretokenized``
990                argument:
991
992                - If ``is_pretokenized=False``: :class:`~tokenizers.TextEncodeInput`
993                - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedEncodeInput`
994
995            is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
996                Whether the input is already pre-tokenized
997
998            add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
999                Whether to add the special tokens
1000
1001        Returns:
1002            A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch
1003        """
1004    def encode_batch_fast(
1005        self, /, input: Sequence[Any], is_pretokenized: bool = False, add_special_tokens: bool = True
1006    ) -> "list[Encoding]":
1007        """
1008        Encode the given batch of inputs. This method is faster than `encode_batch`
1009        because it doesn't keep track of offsets, they will be all zeros.
1010
1011        Example:
1012            Here are some examples of the inputs that are accepted::
1013
1014                encode_batch_fast([
1015                    "A single sequence",
1016                    ("A tuple with a sequence", "And its pair"),
1017                    [ "A", "pre", "tokenized", "sequence" ],
1018                    ([ "A", "pre", "tokenized", "sequence" ], "And its pair")
1019                ])
1020
1021        Args:
1022            input (A :obj:`List`/:obj:`Tuple` of :obj:`~tokenizers.EncodeInput`):
1023                A list of single sequences or pair sequences to encode. Each sequence
1024                can be either raw text or pre-tokenized, according to the ``is_pretokenized``
1025                argument:
1026
1027                - If ``is_pretokenized=False``: :class:`~tokenizers.TextEncodeInput`
1028                - If ``is_pretokenized=True``: :class:`~tokenizers.PreTokenizedEncodeInput`
1029
1030            is_pretokenized (:obj:`bool`, defaults to :obj:`False`):
1031                Whether the input is already pre-tokenized
1032
1033            add_special_tokens (:obj:`bool`, defaults to :obj:`True`):
1034                Whether to add the special tokens
1035
1036        Returns:
1037            A :obj:`List` of :class:`~tokenizers.Encoding`: The encoded batch
1038        """
1039    @property
1040    def encode_special_tokens(self, /) -> bool:
1041        """
1042        Get the value of the `encode_special_tokens` attribute
1043
1044        Returns:
1045            :obj:`bool`: the tokenizer's encode_special_tokens attribute
1046        """
1047    @encode_special_tokens.setter
1048    def encode_special_tokens(self, /, value: bool) -> None:
1049        """
1050        Modifies the tokenizer in order to use or not the special tokens
1051        during encoding.
1052
1053        Args:
1054            value (:obj:`bool`):
1055                Whether to use the special tokens or not
1056        """
1057    @staticmethod
1058    def from_buffer(buffer: bytes) -> "Tokenizer":
1059        """
1060        Instantiate a new :class:`~tokenizers.Tokenizer` from the given buffer.
1061
1062        Args:
1063            buffer (:obj:`bytes`):
1064                A buffer containing a previously serialized :class:`~tokenizers.Tokenizer`
1065
1066        Returns:
1067            :class:`~tokenizers.Tokenizer`: The new tokenizer
1068        """
1069    @staticmethod
1070    def from_file(path: str) -> "Tokenizer":
1071        """
1072        Instantiate a new :class:`~tokenizers.Tokenizer` from the file at the given path.
1073
1074        Args:
1075            path (:obj:`str`):
1076                A path to a local JSON file representing a previously serialized
1077                :class:`~tokenizers.Tokenizer`
1078
1079        Returns:
1080            :class:`~tokenizers.Tokenizer`: The new tokenizer
1081        """
1082    @staticmethod
1083    def from_pretrained(identifier: str, revision: str = ..., token: str | None = None) -> "Tokenizer":
1084        """
1085        Instantiate a new :class:`~tokenizers.Tokenizer` from an existing file on the
1086        Hugging Face Hub.
1087
1088        Args:
1089            identifier (:obj:`str`):
1090                The identifier of a Model on the Hugging Face Hub, that contains
1091                a tokenizer.json file
1092            revision (:obj:`str`, defaults to `main`):
1093                A branch or commit id
1094            token (:obj:`str`, `optional`, defaults to `None`):
1095                An optional auth token used to access private repositories on the
1096                Hugging Face Hub
1097
1098        Returns:
1099            :class:`~tokenizers.Tokenizer`: The new tokenizer
1100        """
1101    @staticmethod
1102    def from_str(json: str) -> "Tokenizer":
1103        """
1104        Instantiate a new :class:`~tokenizers.Tokenizer` from the given JSON string.
1105
1106        Args:
1107            json (:obj:`str`):
1108                A valid JSON string representing a previously serialized
1109                :class:`~tokenizers.Tokenizer`
1110
1111        Returns:
1112            :class:`~tokenizers.Tokenizer`: The new tokenizer
1113        """
1114    def get_added_tokens_decoder(self, /) -> "dict[int, AddedToken]":
1115        """
1116        Get the underlying vocabulary
1117
1118        Returns:
1119            :obj:`Dict[int, AddedToken]`: The vocabulary
1120        """
1121    def get_vocab(self, /, with_added_tokens: bool = True) -> "dict[str, int]":
1122        """
1123        Get the underlying vocabulary
1124
1125        Args:
1126            with_added_tokens (:obj:`bool`, defaults to :obj:`True`):
1127                Whether to include the added tokens
1128
1129        Returns:
1130            :obj:`Dict[str, int]`: The vocabulary
1131        """
1132    def get_vocab_size(self, /, with_added_tokens: bool = True) -> "int":
1133        """
1134        Get the size of the underlying vocabulary
1135
1136        Args:
1137            with_added_tokens (:obj:`bool`, defaults to :obj:`True`):
1138                Whether to include the added tokens
1139
1140        Returns:
1141            :obj:`int`: The size of the vocabulary
1142        """
1143    def id_to_token(self, /, id: int) -> "str | None":
1144        """
1145        Convert the given id to its corresponding token if it exists
1146
1147        Args:
1148            id (:obj:`int`):
1149                The id to convert
1150
1151        Returns:
1152            :obj:`Optional[str]`: An optional token, :obj:`None` if out of vocabulary
1153        """
1154    @property
1155    def model(self, /) -> Any:
1156        """
1157        The :class:`~tokenizers.models.Model` in use by the Tokenizer
1158        """
1159    @model.setter
1160    def model(self, /, model: Model) -> None:
1161        """
1162        Set the :class:`~tokenizers.models.Model`
1163        """
1164    def no_padding(self, /) -> None:
1165        """
1166        Disable padding
1167        """
1168    def no_truncation(self, /) -> None:
1169        """
1170        Disable truncation
1171        """
1172    @property
1173    def normalizer(self, /) -> Any:
1174        """
1175        The `optional` :class:`~tokenizers.normalizers.Normalizer` in use by the Tokenizer
1176        """
1177    @normalizer.setter
1178    def normalizer(self, /, normalizer: Normalizer | None) -> None:
1179        """
1180        Set the :class:`~tokenizers.normalizers.Normalizer`
1181        """
1182    def num_special_tokens_to_add(self, /, is_pair: bool) -> int:
1183        """
1184        Return the number of special tokens that would be added for single/pair sentences.
1185        :param is_pair: Boolean indicating if the input would be a single sentence or a pair
1186        :return:
1187        """
1188    @property
1189    def padding(self, /) -> dict | None:
1190        """
1191        Get the current padding parameters
1192
1193        `Cannot be set, use` :meth:`~tokenizers.Tokenizer.enable_padding` `instead`
1194
1195        Returns:
1196            (:obj:`dict`, `optional`):
1197                A dict with the current padding parameters if padding is enabled
1198        """
1199    def post_process(
1200        self, /, encoding: Encoding, pair: Encoding | None = None, add_special_tokens: bool = True

Showing the first 1,200 of 1329 lines. Download the file for the rest.

codekingpro/portable-devtools · Team Ai