codekingpro/portable-devtools
114k
1"""
2Processors Module
3"""
4
5from _typeshed import Incomplete
6from collections.abc import Sequence as Sequence2
7from tokenizers import Encoding
8from typing import Any, final
9
10@final
11class BertProcessing(PostProcessor):
12 """
13 This post-processor takes care of adding the special tokens needed by
14 a Bert model:
15
16 - a SEP token
17 - a CLS token
18
19 Args:
20 sep (:obj:`Tuple[str, int]`):
21 A tuple with the string representation of the SEP token, and its id
22
23 cls (:obj:`Tuple[str, int]`):
24 A tuple with the string representation of the CLS token, and its id
25
26 Example::
27
28 >>> from tokenizers.processors import BertProcessing
29 >>> processor = BertProcessing(("[SEP]", 102), ("[CLS]", 101))
30 >>> processor.process(encoding)
31 # Encoding with [CLS] at start and [SEP] at end
32 """
33 def __getnewargs__(self, /) -> tuple: ...
34 def __new__(cls, /, sep: tuple[str, int], cls_token: tuple[str, int]) -> BertProcessing: ...
35 @property
36 def cls(self, /) -> tuple: ...
37 @cls.setter
38 def cls(self, /, cls: tuple) -> None: ...
39 @property
40 def sep(self, /) -> tuple: ...
41 @sep.setter
42 def sep(self, /, sep: tuple) -> None: ...
43
44@final
45class ByteLevel(PostProcessor):
46 """
47 This post-processor takes care of trimming the offsets.
48
49 By default, the ByteLevel BPE might include whitespaces in the produced tokens. If you don't
50 want the offsets to include these whitespaces, then this PostProcessor must be used.
51
52 Args:
53 trim_offsets (:obj:`bool`):
54 Whether to trim the whitespaces from the produced offsets.
55
56 add_prefix_space (:obj:`bool`, `optional`, defaults to :obj:`True`):
57 If :obj:`True`, keeps the first token's offset as is. If :obj:`False`, increments
58 the start of the first token's offset by 1. Only has an effect if :obj:`trim_offsets`
59 is set to :obj:`True`.
60
61 Example::
62
63 >>> from tokenizers.processors import ByteLevel
64 >>> processor = ByteLevel(trim_offsets=True)
65 >>> # Offsets will be trimmed to exclude leading whitespace bytes
66 """
67 def __new__(
68 cls,
69 /,
70 add_prefix_space: bool | None = None,
71 trim_offsets: bool | None = None,
72 use_regex: bool | None = None,
73 **_kwargs,
74 ) -> ByteLevel: ...
75 @property
76 def add_prefix_space(self, /) -> bool: ...
77 @add_prefix_space.setter
78 def add_prefix_space(self, /, add_prefix_space: bool) -> None: ...
79 @property
80 def trim_offsets(self, /) -> bool: ...
81 @trim_offsets.setter
82 def trim_offsets(self, /, trim_offsets: bool) -> None: ...
83 @property
84 def use_regex(self, /) -> bool: ...
85 @use_regex.setter
86 def use_regex(self, /, use_regex: bool) -> None: ...
87
88class PostProcessor:
89 """
90 Base class for all post-processors
91
92 This class is not supposed to be instantiated directly. Instead, any implementation of
93 a PostProcessor will return an instance of this class when instantiated.
94 """
95 def __getstate__(self, /) -> Any: ...
96 def __repr__(self, /) -> str: ...
97 def __setstate__(self, /, state: Any) -> None: ...
98 def __str__(self, /) -> str: ...
99 def num_special_tokens_to_add(self, /, is_pair: bool) -> int:
100 """
101 Return the number of special tokens that would be added for single/pair sentences.
102
103 Args:
104 is_pair (:obj:`bool`):
105 Whether the input would be a pair of sequences
106
107 Returns:
108 :obj:`int`: The number of tokens to add
109 """
110 def process(
111 self, /, encoding: Encoding, pair: Encoding | None = None, add_special_tokens: bool = True
112 ) -> "Encoding":
113 """
114 Post-process the given encodings, generating the final one
115
116 Args:
117 encoding (:class:`~tokenizers.Encoding`):
118 The encoding for the first sequence
119
120 pair (:class:`~tokenizers.Encoding`, `optional`):
121 The encoding for the pair sequence
122
123 add_special_tokens (:obj:`bool`):
124 Whether to add the special tokens
125
126 Return:
127 :class:`~tokenizers.Encoding`: The final encoding
128 """
129
130@final
131class RobertaProcessing(PostProcessor):
132 """
133 This post-processor takes care of adding the special tokens needed by
134 a Roberta model:
135
136 - a SEP token
137 - a CLS token
138
139 It also takes care of trimming the offsets.
140 By default, the ByteLevel BPE might include whitespaces in the produced tokens. If you don't
141 want the offsets to include these whitespaces, then this PostProcessor should be initialized
142 with :obj:`trim_offsets=True`
143
144 Args:
145 sep (:obj:`Tuple[str, int]`):
146 A tuple with the string representation of the SEP token, and its id
147
148 cls (:obj:`Tuple[str, int]`):
149 A tuple with the string representation of the CLS token, and its id
150
151 trim_offsets (:obj:`bool`, `optional`, defaults to :obj:`True`):
152 Whether to trim the whitespaces from the produced offsets.
153
154 add_prefix_space (:obj:`bool`, `optional`, defaults to :obj:`True`):
155 Whether the add_prefix_space option was enabled during pre-tokenization. This
156 is relevant because it defines the way the offsets are trimmed out.
157
158 Example::
159
160 >>> from tokenizers.processors import RobertaProcessing
161 >>> processor = RobertaProcessing(("</s>", 2), ("<s>", 0))
162 >>> processor.process(encoding)
163 # Encoding with <s> at start and </s> at end
164 """
165 def __getnewargs__(self, /) -> tuple: ...
166 def __new__(
167 cls,
168 /,
169 sep: tuple[str, int],
170 cls_token: tuple[str, int],
171 trim_offsets: bool = True,
172 add_prefix_space: bool = True,
173 ) -> RobertaProcessing: ...
174 @property
175 def add_prefix_space(self, /) -> bool: ...
176 @add_prefix_space.setter
177 def add_prefix_space(self, /, add_prefix_space: bool) -> None: ...
178 @property
179 def cls(self, /) -> tuple: ...
180 @cls.setter
181 def cls(self, /, cls: tuple) -> None: ...
182 @property
183 def sep(self, /) -> tuple: ...
184 @sep.setter
185 def sep(self, /, sep: tuple) -> None: ...
186 @property
187 def trim_offsets(self, /) -> bool: ...
188 @trim_offsets.setter
189 def trim_offsets(self, /, trim_offsets: bool) -> None: ...
190
191@final
192class Sequence(PostProcessor):
193 """
194 Sequence Processor
195
196 Chains multiple post-processors together, applying them in order. Each processor
197 in the sequence processes the output of the previous one.
198
199 Args:
200 processors (:obj:`List[PostProcessor]`):
201 The list of post-processors to chain together.
202
203 Example::
204
205 >>> from tokenizers.processors import BertProcessing, ByteLevel, Sequence
206 >>> processor = Sequence([ByteLevel(trim_offsets=True), BertProcessing(("[SEP]", 102), ("[CLS]", 101))])
207 """
208 def __getitem__(self, /, index: int) -> Any: ...
209 def __getnewargs__(self, /) -> tuple: ...
210 def __new__(cls, /, processors_py: list) -> Sequence: ...
211 def __setitem__(self, /, index: int, value: Any) -> None: ...
212
213@final
214class TemplateProcessing(PostProcessor):
215 """
216 Provides a way to specify templates in order to add the special tokens to each
217 input sequence as relevant.
218
219 Let's take :obj:`BERT` tokenizer as an example. It uses two special tokens, used to
220 delimitate each sequence. :obj:`[CLS]` is always used at the beginning of the first
221 sequence, and :obj:`[SEP]` is added at the end of both the first, and the pair
222 sequences. The final result looks like this:
223
224 - Single sequence: :obj:`[CLS] Hello there [SEP]`
225 - Pair sequences: :obj:`[CLS] My name is Anthony [SEP] What is my name? [SEP]`
226
227 With the type ids as following::
228
229 [CLS] ... [SEP] ... [SEP]
230 0 0 0 1 1
231
232 You can achieve such behavior using a TemplateProcessing::
233
234 TemplateProcessing(
235 single="[CLS] $0 [SEP]",
236 pair="[CLS] $A [SEP] $B:1 [SEP]:1",
237 special_tokens=[("[CLS]", 1), ("[SEP]", 0)],
238 )
239
240 In this example, each input sequence is identified using a ``$`` construct. This identifier
241 lets us specify each input sequence, and the type_id to use. When nothing is specified,
242 it uses the default values. Here are the different ways to specify it:
243
244 - Specifying the sequence, with default ``type_id == 0``: ``$A`` or ``$B``
245 - Specifying the `type_id` with default ``sequence == A``: ``$0``, ``$1``, ``$2``, ...
246 - Specifying both: ``$A:0``, ``$B:1``, ...
247
248 The same construct is used for special tokens: ``<identifier>(:<type_id>)?``.
249
250 **Warning**: You must ensure that you are giving the correct tokens/ids as these
251 will be added to the Encoding without any further check. If the given ids correspond
252 to something totally different in a `Tokenizer` using this `PostProcessor`, it
253 might lead to unexpected results.
254
255 Args:
256 single (:obj:`Template`):
257 The template used for single sequences
258
259 pair (:obj:`Template`):
260 The template used when both sequences are specified
261
262 special_tokens (:obj:`Tokens`):
263 The list of special tokens used in each sequences
264
265 Types:
266
267 Template (:obj:`str` or :obj:`List`):
268 - If a :obj:`str` is provided, the whitespace is used as delimiter between tokens
269 - If a :obj:`List[str]` is provided, a list of tokens
270
271 Tokens (:obj:`List[Union[Tuple[int, str], Tuple[str, int], dict]]`):
272 - A :obj:`Tuple` with both a token and its associated ID, in any order
273 - A :obj:`dict` with the following keys:
274 - "id": :obj:`str` => The special token id, as specified in the Template
275 - "ids": :obj:`List[int]` => The associated IDs
276 - "tokens": :obj:`List[str]` => The associated tokens
277
278 The given dict expects the provided :obj:`ids` and :obj:`tokens` lists to have
279 the same length.
280 """
281 def __new__(
282 cls,
283 /,
284 single: Incomplete | None = None,
285 pair: Incomplete | None = None,
286 special_tokens: Sequence2[Incomplete] | None = None,
287 ) -> TemplateProcessing: ...
288 @property
289 def single(self, /) -> str: ...
290 @single.setter
291 def single(self, /, single: Incomplete) -> None: ...
292 