Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
processors.pyi292 linesDownload Raw Back to tokenizers
1"""
2Processors Module
3"""
4
5from _typeshed import Incomplete
6from collections.abc import Sequence as Sequence2
7from tokenizers import Encoding
8from typing import Any, final
9
10@final
11class BertProcessing(PostProcessor):
12    """
13    This post-processor takes care of adding the special tokens needed by
14    a Bert model:
15
16        - a SEP token
17        - a CLS token
18
19    Args:
20        sep (:obj:`Tuple[str, int]`):
21            A tuple with the string representation of the SEP token, and its id
22
23        cls (:obj:`Tuple[str, int]`):
24            A tuple with the string representation of the CLS token, and its id
25
26    Example::
27
28        >>> from tokenizers.processors import BertProcessing
29        >>> processor = BertProcessing(("[SEP]", 102), ("[CLS]", 101))
30        >>> processor.process(encoding)
31        # Encoding with [CLS] at start and [SEP] at end
32    """
33    def __getnewargs__(self, /) -> tuple: ...
34    def __new__(cls, /, sep: tuple[str, int], cls_token: tuple[str, int]) -> BertProcessing: ...
35    @property
36    def cls(self, /) -> tuple: ...
37    @cls.setter
38    def cls(self, /, cls: tuple) -> None: ...
39    @property
40    def sep(self, /) -> tuple: ...
41    @sep.setter
42    def sep(self, /, sep: tuple) -> None: ...
43
44@final
45class ByteLevel(PostProcessor):
46    """
47    This post-processor takes care of trimming the offsets.
48
49    By default, the ByteLevel BPE might include whitespaces in the produced tokens. If you don't
50    want the offsets to include these whitespaces, then this PostProcessor must be used.
51
52    Args:
53        trim_offsets (:obj:`bool`):
54            Whether to trim the whitespaces from the produced offsets.
55
56        add_prefix_space (:obj:`bool`, `optional`, defaults to :obj:`True`):
57            If :obj:`True`, keeps the first token's offset as is. If :obj:`False`, increments
58            the start of the first token's offset by 1. Only has an effect if :obj:`trim_offsets`
59            is set to :obj:`True`.
60
61    Example::
62
63        >>> from tokenizers.processors import ByteLevel
64        >>> processor = ByteLevel(trim_offsets=True)
65        >>> # Offsets will be trimmed to exclude leading whitespace bytes
66    """
67    def __new__(
68        cls,
69        /,
70        add_prefix_space: bool | None = None,
71        trim_offsets: bool | None = None,
72        use_regex: bool | None = None,
73        **_kwargs,
74    ) -> ByteLevel: ...
75    @property
76    def add_prefix_space(self, /) -> bool: ...
77    @add_prefix_space.setter
78    def add_prefix_space(self, /, add_prefix_space: bool) -> None: ...
79    @property
80    def trim_offsets(self, /) -> bool: ...
81    @trim_offsets.setter
82    def trim_offsets(self, /, trim_offsets: bool) -> None: ...
83    @property
84    def use_regex(self, /) -> bool: ...
85    @use_regex.setter
86    def use_regex(self, /, use_regex: bool) -> None: ...
87
88class PostProcessor:
89    """
90    Base class for all post-processors
91
92    This class is not supposed to be instantiated directly. Instead, any implementation of
93    a PostProcessor will return an instance of this class when instantiated.
94    """
95    def __getstate__(self, /) -> Any: ...
96    def __repr__(self, /) -> str: ...
97    def __setstate__(self, /, state: Any) -> None: ...
98    def __str__(self, /) -> str: ...
99    def num_special_tokens_to_add(self, /, is_pair: bool) -> int:
100        """
101        Return the number of special tokens that would be added for single/pair sentences.
102
103        Args:
104            is_pair (:obj:`bool`):
105                Whether the input would be a pair of sequences
106
107        Returns:
108            :obj:`int`: The number of tokens to add
109        """
110    def process(
111        self, /, encoding: Encoding, pair: Encoding | None = None, add_special_tokens: bool = True
112    ) -> "Encoding":
113        """
114        Post-process the given encodings, generating the final one
115
116        Args:
117            encoding (:class:`~tokenizers.Encoding`):
118                The encoding for the first sequence
119
120            pair (:class:`~tokenizers.Encoding`, `optional`):
121                The encoding for the pair sequence
122
123            add_special_tokens (:obj:`bool`):
124                Whether to add the special tokens
125
126        Return:
127            :class:`~tokenizers.Encoding`: The final encoding
128        """
129
130@final
131class RobertaProcessing(PostProcessor):
132    """
133    This post-processor takes care of adding the special tokens needed by
134    a Roberta model:
135
136        - a SEP token
137        - a CLS token
138
139    It also takes care of trimming the offsets.
140    By default, the ByteLevel BPE might include whitespaces in the produced tokens. If you don't
141    want the offsets to include these whitespaces, then this PostProcessor should be initialized
142    with :obj:`trim_offsets=True`
143
144    Args:
145        sep (:obj:`Tuple[str, int]`):
146            A tuple with the string representation of the SEP token, and its id
147
148        cls (:obj:`Tuple[str, int]`):
149            A tuple with the string representation of the CLS token, and its id
150
151        trim_offsets (:obj:`bool`, `optional`, defaults to :obj:`True`):
152            Whether to trim the whitespaces from the produced offsets.
153
154        add_prefix_space (:obj:`bool`, `optional`, defaults to :obj:`True`):
155            Whether the add_prefix_space option was enabled during pre-tokenization. This
156            is relevant because it defines the way the offsets are trimmed out.
157
158    Example::
159
160        >>> from tokenizers.processors import RobertaProcessing
161        >>> processor = RobertaProcessing(("</s>", 2), ("<s>", 0))
162        >>> processor.process(encoding)
163        # Encoding with <s> at start and </s> at end
164    """
165    def __getnewargs__(self, /) -> tuple: ...
166    def __new__(
167        cls,
168        /,
169        sep: tuple[str, int],
170        cls_token: tuple[str, int],
171        trim_offsets: bool = True,
172        add_prefix_space: bool = True,
173    ) -> RobertaProcessing: ...
174    @property
175    def add_prefix_space(self, /) -> bool: ...
176    @add_prefix_space.setter
177    def add_prefix_space(self, /, add_prefix_space: bool) -> None: ...
178    @property
179    def cls(self, /) -> tuple: ...
180    @cls.setter
181    def cls(self, /, cls: tuple) -> None: ...
182    @property
183    def sep(self, /) -> tuple: ...
184    @sep.setter
185    def sep(self, /, sep: tuple) -> None: ...
186    @property
187    def trim_offsets(self, /) -> bool: ...
188    @trim_offsets.setter
189    def trim_offsets(self, /, trim_offsets: bool) -> None: ...
190
191@final
192class Sequence(PostProcessor):
193    """
194    Sequence Processor
195
196    Chains multiple post-processors together, applying them in order. Each processor
197    in the sequence processes the output of the previous one.
198
199    Args:
200        processors (:obj:`List[PostProcessor]`):
201            The list of post-processors to chain together.
202
203    Example::
204
205        >>> from tokenizers.processors import BertProcessing, ByteLevel, Sequence
206        >>> processor = Sequence([ByteLevel(trim_offsets=True), BertProcessing(("[SEP]", 102), ("[CLS]", 101))])
207    """
208    def __getitem__(self, /, index: int) -> Any: ...
209    def __getnewargs__(self, /) -> tuple: ...
210    def __new__(cls, /, processors_py: list) -> Sequence: ...
211    def __setitem__(self, /, index: int, value: Any) -> None: ...
212
213@final
214class TemplateProcessing(PostProcessor):
215    """
216    Provides a way to specify templates in order to add the special tokens to each
217    input sequence as relevant.
218
219    Let's take :obj:`BERT` tokenizer as an example. It uses two special tokens, used to
220    delimitate each sequence. :obj:`[CLS]` is always used at the beginning of the first
221    sequence, and :obj:`[SEP]` is added at the end of both the first, and the pair
222    sequences. The final result looks like this:
223
224        - Single sequence: :obj:`[CLS] Hello there [SEP]`
225        - Pair sequences: :obj:`[CLS] My name is Anthony [SEP] What is my name? [SEP]`
226
227    With the type ids as following::
228
229        [CLS]   ...   [SEP]   ...   [SEP]
230          0      0      0      1      1
231
232    You can achieve such behavior using a TemplateProcessing::
233
234        TemplateProcessing(
235            single="[CLS] $0 [SEP]",
236            pair="[CLS] $A [SEP] $B:1 [SEP]:1",
237            special_tokens=[("[CLS]", 1), ("[SEP]", 0)],
238        )
239
240    In this example, each input sequence is identified using a ``$`` construct. This identifier
241    lets us specify each input sequence, and the type_id to use. When nothing is specified,
242    it uses the default values. Here are the different ways to specify it:
243
244        - Specifying the sequence, with default ``type_id == 0``: ``$A`` or ``$B``
245        - Specifying the `type_id` with default ``sequence == A``: ``$0``, ``$1``, ``$2``, ...
246        - Specifying both: ``$A:0``, ``$B:1``, ...
247
248    The same construct is used for special tokens: ``<identifier>(:<type_id>)?``.
249
250    **Warning**: You must ensure that you are giving the correct tokens/ids as these
251    will be added to the Encoding without any further check. If the given ids correspond
252    to something totally different in a `Tokenizer` using this `PostProcessor`, it
253    might lead to unexpected results.
254
255    Args:
256        single (:obj:`Template`):
257            The template used for single sequences
258
259        pair (:obj:`Template`):
260            The template used when both sequences are specified
261
262        special_tokens (:obj:`Tokens`):
263            The list of special tokens used in each sequences
264
265    Types:
266
267        Template (:obj:`str` or :obj:`List`):
268            - If a :obj:`str` is provided, the whitespace is used as delimiter between tokens
269            - If a :obj:`List[str]` is provided, a list of tokens
270
271        Tokens (:obj:`List[Union[Tuple[int, str], Tuple[str, int], dict]]`):
272            - A :obj:`Tuple` with both a token and its associated ID, in any order
273            - A :obj:`dict` with the following keys:
274                - "id": :obj:`str` => The special token id, as specified in the Template
275                - "ids": :obj:`List[int]` => The associated IDs
276                - "tokens": :obj:`List[str]` => The associated tokens
277
278             The given dict expects the provided :obj:`ids` and :obj:`tokens` lists to have
279             the same length.
280    """
281    def __new__(
282        cls,
283        /,
284        single: Incomplete | None = None,
285        pair: Incomplete | None = None,
286        special_tokens: Sequence2[Incomplete] | None = None,
287    ) -> TemplateProcessing: ...
288    @property
289    def single(self, /) -> str: ...
290    @single.setter
291    def single(self, /, single: Incomplete) -> None: ...
292 
codekingpro/portable-devtools · Team Ai