codekingpro/portable-devtools
114k
1"""
2Normalizers Module
3"""
4
5from collections.abc import Sequence as Sequence2
6from tokenizers import NormalizedString, Regex
7from typing import Any, final
8
9@final
10class BertNormalizer(Normalizer):
11 """
12 BertNormalizer
13
14 Takes care of normalizing raw text before giving it to a Bert model.
15 This includes cleaning the text, handling accents, chinese chars and lowercasing
16
17 Args:
18 clean_text (:obj:`bool`, `optional`, defaults to :obj:`True`):
19 Whether to clean the text, by removing any control characters
20 and replacing all whitespaces by the classic one.
21
22 handle_chinese_chars (:obj:`bool`, `optional`, defaults to :obj:`True`):
23 Whether to handle chinese chars by putting spaces around them.
24
25 strip_accents (:obj:`bool`, `optional`):
26 Whether to strip all accents. If this option is not specified (ie == None),
27 then it will be determined by the value for `lowercase` (as in the original Bert).
28
29 lowercase (:obj:`bool`, `optional`, defaults to :obj:`True`):
30 Whether to lowercase.
31
32 Example::
33
34 >>> from tokenizers.normalizers import BertNormalizer
35 >>> normalizer = BertNormalizer(lowercase=True)
36 >>> normalizer.normalize_str("Héllo WORLD")
37 'hello world'
38 """
39 def __new__(
40 cls,
41 /,
42 clean_text: bool = True,
43 handle_chinese_chars: bool = True,
44 strip_accents: bool | None = None,
45 lowercase: bool = True,
46 ) -> BertNormalizer: ...
47 @property
48 def clean_text(self, /) -> bool: ...
49 @clean_text.setter
50 def clean_text(self, /, clean_text: bool) -> None: ...
51 @property
52 def handle_chinese_chars(self, /) -> bool: ...
53 @handle_chinese_chars.setter
54 def handle_chinese_chars(self, /, handle_chinese_chars: bool) -> None: ...
55 @property
56 def lowercase(self, /) -> bool: ...
57 @lowercase.setter
58 def lowercase(self, /, lowercase: bool) -> None: ...
59 @property
60 def strip_accents(self, /) -> bool | None: ...
61 @strip_accents.setter
62 def strip_accents(self, /, strip_accents: bool | None) -> None: ...
63
64@final
65class ByteLevel(Normalizer):
66 """
67 Bytelevel Normalizer
68
69 Converts all bytes in the input to their Unicode representation using the GPT-2
70 byte-to-unicode mapping. Every byte value (0–255) is mapped to a unique visible
71 character so that any arbitrary binary input can be tokenized without needing a
72 special unknown token.
73
74 This normalizer is used together with the
75 :class:`~tokenizers.pre_tokenizers.ByteLevel` pre-tokenizer and
76 :class:`~tokenizers.decoders.ByteLevel` decoder.
77
78 Example::
79
80 >>> from tokenizers.normalizers import ByteLevel
81 >>> normalizer = ByteLevel()
82 >>> normalizer.normalize_str("hello\nworld")
83 'helloĊworld'
84 """
85 def __new__(cls, /) -> ByteLevel: ...
86
87@final
88class Lowercase(Normalizer):
89 """
90 Lowercase Normalizer
91
92 Converts all text to lowercase using Unicode-aware lowercasing. This is equivalent
93 to calling :meth:`str.lower` on the input.
94
95 Example::
96
97 >>> from tokenizers.normalizers import Lowercase
98 >>> normalizer = Lowercase()
99 >>> normalizer.normalize_str("Hello World")
100 'hello world'
101 """
102 def __new__(cls, /) -> Lowercase: ...
103
104@final
105class NFC(Normalizer):
106 """
107 NFC Unicode Normalizer
108
109 Applies Unicode NFC (Canonical Decomposition, followed by Canonical Composition)
110 normalization. First decomposes characters, then recomposes them using canonical
111 composition rules. This produces the canonical composed form.
112
113 Example::
114
115 >>> from tokenizers.normalizers import NFC
116 >>> normalizer = NFC()
117 >>> normalizer.normalize_str("e\u0301") # 'e' + combining accent
118 'é'
119 """
120 def __new__(cls, /) -> NFC: ...
121
122@final
123class NFD(Normalizer):
124 """
125 NFD Unicode Normalizer
126
127 Applies Unicode NFD (Canonical Decomposition) normalization. Decomposes characters into
128 their canonical components. For example, accented characters like ``é`` (U+00E9) are
129 decomposed into ``e`` (U+0065) + combining accent (U+0301).
130
131 This is often used as a first step before stripping accents with
132 :class:`~tokenizers.normalizers.StripAccents`.
133
134 Example::
135
136 >>> from tokenizers.normalizers import NFD
137 >>> normalizer = NFD()
138 >>> normalizer.normalize_str("Héllo")
139 'He\u0301llo'
140 """
141 def __new__(cls, /) -> NFD: ...
142
143@final
144class NFKC(Normalizer):
145 """
146 NFKC Unicode Normalizer
147
148 Applies Unicode NFKC (Compatibility Decomposition, followed by Canonical Composition)
149 normalization. Like NFC but also maps compatibility characters to their canonical
150 equivalents. This is the normalization used by Python's :func:`str.casefold` and
151 by many NLP pipelines.
152
153 Example::
154
155 >>> from tokenizers.normalizers import NFKC
156 >>> normalizer = NFKC()
157 >>> normalizer.normalize_str("fine caf\u00e9")
158 'fine café'
159 """
160 def __new__(cls, /) -> NFKC: ...
161
162@final
163class NFKD(Normalizer):
164 """
165 NFKD Unicode Normalizer
166
167 Applies Unicode NFKD (Compatibility Decomposition) normalization. Like NFD but also
168 decomposes compatibility characters. For example, the ligature ``fi`` (U+FB01) is
169 decomposed into ``f`` + ``i``.
170
171 Example::
172
173 >>> from tokenizers.normalizers import NFKD
174 >>> normalizer = NFKD()
175 >>> normalizer.normalize_str("fine")
176 'fine'
177 """
178 def __new__(cls, /) -> NFKD: ...
179
180@final
181class Nmt(Normalizer):
182 """
183 Nmt normalizer
184
185 Normalizer used in the Google NMT pipeline. It handles various text cleaning tasks
186 including removing control characters, normalizing whitespace, and replacing certain
187 Unicode characters. This is equivalent to the normalization done in the original
188 SentencePiece NMT preprocessing.
189
190 Example::
191
192 >>> from tokenizers.normalizers import Nmt
193 >>> normalizer = Nmt()
194 >>> normalizer.normalize_str("Hello\x00World")
195 'Hello World'
196 """
197 def __new__(cls, /) -> Nmt: ...
198
199class Normalizer:
200 """
201 Base class for all normalizers
202
203 This class is not supposed to be instantiated directly. Instead, any implementation of a
204 Normalizer will return an instance of this class when instantiated.
205 """
206 def __getstate__(self, /) -> Any: ...
207 def __repr__(self, /) -> str: ...
208 def __setstate__(self, /, state: Any) -> None: ...
209 def __str__(self, /) -> str: ...
210 @staticmethod
211 def custom(obj: Any) -> Normalizer: ...
212 def normalize(self, /, normalized: NormalizedString | Any) -> None:
213 """
214 Normalize a :class:`~tokenizers.NormalizedString` in-place
215
216 This method allows to modify a :class:`~tokenizers.NormalizedString` to
217 keep track of the alignment information. If you just want to see the result
218 of the normalization on a raw string, you can use
219 :meth:`~tokenizers.normalizers.Normalizer.normalize_str`
220
221 Args:
222 normalized (:class:`~tokenizers.NormalizedString`):
223 The normalized string on which to apply this
224 :class:`~tokenizers.normalizers.Normalizer`
225 """
226 def normalize_str(self, /, sequence: str) -> str:
227 """
228 Normalize the given string
229
230 This method provides a way to visualize the effect of a
231 :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment
232 information. If you need to get/convert offsets, you can use
233 :meth:`~tokenizers.normalizers.Normalizer.normalize`
234
235 Args:
236 sequence (:obj:`str`):
237 A string to normalize
238
239 Returns:
240 :obj:`str`: A string after normalization
241 """
242
243@final
244class Precompiled(Normalizer):
245 """
246 Precompiled normalizer
247
248 A normalizer that uses a precompiled character map built from a SentencePiece model.
249 This normalizer is automatically extracted from SentencePiece ``.model`` files and
250 should not be constructed manually — it is used internally for full compatibility
251 with SentencePiece-based tokenizers.
252
253 Args:
254 precompiled_charsmap (:obj:`bytes`):
255 The raw bytes of the precompiled character map, as found inside a
256 SentencePiece ``.model`` file.
257 """
258 def __new__(cls, /, precompiled_charsmap: Sequence2[int]) -> Precompiled: ...
259
260@final
261class Prepend(Normalizer):
262 """
263 Prepend normalizer
264
265 Prepends a given string to the beginning of the input. This is typically used to
266 add a meta-symbol such as ``▁`` (U+2581) at the start of each sequence, which is
267 the convention used by SentencePiece-based models to indicate that a token appears
268 at the start of a word.
269
270 Args:
271 prepend (:obj:`str`, defaults to :obj:`"▁"`):
272 The string to prepend to the input.
273
274 Example::
275
276 >>> from tokenizers.normalizers import Prepend
277 >>> normalizer = Prepend("▁")
278 >>> normalizer.normalize_str("hello")
279 '▁hello'
280 """
281 def __new__(cls, /, prepend: str = ...) -> Prepend: ...
282 @property
283 def prepend(self, /) -> str: ...
284 @prepend.setter
285 def prepend(self, /, prepend: str) -> None: ...
286
287@final
288class Replace(Normalizer):
289 """
290 Replace normalizer
291
292 Replaces occurrences of a pattern in the input string with the given content.
293 The pattern can be either a plain string or a regular expression wrapped in
294 :class:`~tokenizers.Regex`.
295
296 Args:
297 pattern (:obj:`str` or :class:`~tokenizers.Regex`):
298 The pattern to search for. Use a plain string for literal replacement,
299 or wrap a regex pattern in :class:`~tokenizers.Regex` for regex replacement.
300
301 content (:obj:`str`):
302 The string to replace each match with.
303
304 Example::
305
306 >>> from tokenizers import Regex
307 >>> from tokenizers.normalizers import Replace
308 >>> # Replace a literal string
309 >>> Replace(".", " ").normalize_str("hello.world")
310 'hello world'
311 >>> # Replace using a regex
312 >>> Replace(Regex(r"\s+"), " ").normalize_str("hello world")
313 'hello world'
314 """
315 def __new__(cls, /, pattern: str | Regex, content: str) -> Replace: ...
316 @property
317 def content(self, /) -> str: ...
318 @content.setter
319 def content(self, /, content: str) -> None: ...
320 @property
321 def pattern(self, /) -> None: ...
322 @pattern.setter
323 def pattern(self, /, _pattern: str | Regex) -> None: ...
324
325@final
326class Sequence(Normalizer):
327 """
328 Allows concatenating multiple other Normalizer as a Sequence.
329 All the normalizers run in sequence in the given order
330
331 Args:
332 normalizers (:obj:`List[Normalizer]`):
333 A list of Normalizer to be run as a sequence
334
335 Example::
336
337 >>> from tokenizers.normalizers import NFD, Lowercase, StripAccents, Sequence
338 >>> normalizer = Sequence([NFD(), Lowercase(), StripAccents()])
339 >>> normalizer.normalize_str("Héllo Wörld")
340 'hello world'
341 """
342 def __getitem__(self, /, index: int) -> Any: ...
343 def __getnewargs__(self, /) -> tuple: ...
344 def __len__(self, /) -> int: ...
345 def __new__(cls, /, normalizers: list) -> Sequence: ...
346 def __setitem__(self, /, index: int, value: Any) -> None: ...
347
348@final
349class Strip(Normalizer):
350 """
351 Strip normalizer
352
353 Removes leading and/or trailing whitespace from the input string.
354
355 Args:
356 left (:obj:`bool`, defaults to :obj:`True`):
357 Whether to strip leading (left) whitespace.
358
359 right (:obj:`bool`, defaults to :obj:`True`):
360 Whether to strip trailing (right) whitespace.
361
362 Example::
363
364 >>> from tokenizers.normalizers import Strip
365 >>> normalizer = Strip()
366 >>> normalizer.normalize_str(" hello world ")
367 'hello world'
368 >>> Strip(right=False).normalize_str(" hello ")
369 'hello '
370 """
371 def __new__(cls, /, left: bool = True, right: bool = True) -> Strip: ...
372 @property
373 def left(self, /) -> bool: ...
374 @left.setter
375 def left(self, /, left: bool) -> None: ...
376 @property
377 def right(self, /) -> bool: ...
378 @right.setter
379 def right(self, /, right: bool) -> None: ...
380
381@final
382class StripAccents(Normalizer):
383 """
384 StripAccents normalizer
385
386 Strips all accent marks (combining diacritical characters) from the input. This
387 normalizer should typically be used after applying :class:`~tokenizers.normalizers.NFD`
388 or :class:`~tokenizers.normalizers.NFKD` decomposition, which separates base
389 characters from their combining accents.
390
391 Example::
392
393 >>> from tokenizers.normalizers import NFD, StripAccents, Sequence
394 >>> normalizer = Sequence([NFD(), StripAccents()])
395 >>> normalizer.normalize_str("café")
396 'cafe'
397 """
398 def __new__(cls, /) -> StripAccents: ...
399 