Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes14kdownloads
normalizers.pyi399 linesDownload Raw Back to tokenizers
1"""
2Normalizers Module
3"""
4
5from collections.abc import Sequence as Sequence2
6from tokenizers import NormalizedString, Regex
7from typing import Any, final
8
9@final
10class BertNormalizer(Normalizer):
11    """
12    BertNormalizer
13
14    Takes care of normalizing raw text before giving it to a Bert model.
15    This includes cleaning the text, handling accents, chinese chars and lowercasing
16
17    Args:
18        clean_text (:obj:`bool`, `optional`, defaults to :obj:`True`):
19            Whether to clean the text, by removing any control characters
20            and replacing all whitespaces by the classic one.
21
22        handle_chinese_chars (:obj:`bool`, `optional`, defaults to :obj:`True`):
23            Whether to handle chinese chars by putting spaces around them.
24
25        strip_accents (:obj:`bool`, `optional`):
26            Whether to strip all accents. If this option is not specified (ie == None),
27            then it will be determined by the value for `lowercase` (as in the original Bert).
28
29        lowercase (:obj:`bool`, `optional`, defaults to :obj:`True`):
30            Whether to lowercase.
31
32    Example::
33
34        >>> from tokenizers.normalizers import BertNormalizer
35        >>> normalizer = BertNormalizer(lowercase=True)
36        >>> normalizer.normalize_str("Héllo WORLD")
37        'hello world'
38    """
39    def __new__(
40        cls,
41        /,
42        clean_text: bool = True,
43        handle_chinese_chars: bool = True,
44        strip_accents: bool | None = None,
45        lowercase: bool = True,
46    ) -> BertNormalizer: ...
47    @property
48    def clean_text(self, /) -> bool: ...
49    @clean_text.setter
50    def clean_text(self, /, clean_text: bool) -> None: ...
51    @property
52    def handle_chinese_chars(self, /) -> bool: ...
53    @handle_chinese_chars.setter
54    def handle_chinese_chars(self, /, handle_chinese_chars: bool) -> None: ...
55    @property
56    def lowercase(self, /) -> bool: ...
57    @lowercase.setter
58    def lowercase(self, /, lowercase: bool) -> None: ...
59    @property
60    def strip_accents(self, /) -> bool | None: ...
61    @strip_accents.setter
62    def strip_accents(self, /, strip_accents: bool | None) -> None: ...
63
64@final
65class ByteLevel(Normalizer):
66    """
67    Bytelevel Normalizer
68
69    Converts all bytes in the input to their Unicode representation using the GPT-2
70    byte-to-unicode mapping. Every byte value (0–255) is mapped to a unique visible
71    character so that any arbitrary binary input can be tokenized without needing a
72    special unknown token.
73
74    This normalizer is used together with the
75    :class:`~tokenizers.pre_tokenizers.ByteLevel` pre-tokenizer and
76    :class:`~tokenizers.decoders.ByteLevel` decoder.
77
78    Example::
79
80        >>> from tokenizers.normalizers import ByteLevel
81        >>> normalizer = ByteLevel()
82        >>> normalizer.normalize_str("hello\nworld")
83        'helloĊworld'
84    """
85    def __new__(cls, /) -> ByteLevel: ...
86
87@final
88class Lowercase(Normalizer):
89    """
90    Lowercase Normalizer
91
92    Converts all text to lowercase using Unicode-aware lowercasing. This is equivalent
93    to calling :meth:`str.lower` on the input.
94
95    Example::
96
97        >>> from tokenizers.normalizers import Lowercase
98        >>> normalizer = Lowercase()
99        >>> normalizer.normalize_str("Hello World")
100        'hello world'
101    """
102    def __new__(cls, /) -> Lowercase: ...
103
104@final
105class NFC(Normalizer):
106    """
107    NFC Unicode Normalizer
108
109    Applies Unicode NFC (Canonical Decomposition, followed by Canonical Composition)
110    normalization. First decomposes characters, then recomposes them using canonical
111    composition rules. This produces the canonical composed form.
112
113    Example::
114
115        >>> from tokenizers.normalizers import NFC
116        >>> normalizer = NFC()
117        >>> normalizer.normalize_str("e\u0301")  # 'e' + combining accent
118        'é'
119    """
120    def __new__(cls, /) -> NFC: ...
121
122@final
123class NFD(Normalizer):
124    """
125    NFD Unicode Normalizer
126
127    Applies Unicode NFD (Canonical Decomposition) normalization. Decomposes characters into
128    their canonical components. For example, accented characters like ``é`` (U+00E9) are
129    decomposed into ``e`` (U+0065) + combining accent (U+0301).
130
131    This is often used as a first step before stripping accents with
132    :class:`~tokenizers.normalizers.StripAccents`.
133
134    Example::
135
136        >>> from tokenizers.normalizers import NFD
137        >>> normalizer = NFD()
138        >>> normalizer.normalize_str("Héllo")
139        'He\u0301llo'
140    """
141    def __new__(cls, /) -> NFD: ...
142
143@final
144class NFKC(Normalizer):
145    """
146    NFKC Unicode Normalizer
147
148    Applies Unicode NFKC (Compatibility Decomposition, followed by Canonical Composition)
149    normalization. Like NFC but also maps compatibility characters to their canonical
150    equivalents. This is the normalization used by Python's :func:`str.casefold` and
151    by many NLP pipelines.
152
153    Example::
154
155        >>> from tokenizers.normalizers import NFKC
156        >>> normalizer = NFKC()
157        >>> normalizer.normalize_str("fine caf\u00e9")
158        'fine café'
159    """
160    def __new__(cls, /) -> NFKC: ...
161
162@final
163class NFKD(Normalizer):
164    """
165    NFKD Unicode Normalizer
166
167    Applies Unicode NFKD (Compatibility Decomposition) normalization. Like NFD but also
168    decomposes compatibility characters. For example, the ligature ``fi`` (U+FB01) is
169    decomposed into ``f`` + ``i``.
170
171    Example::
172
173        >>> from tokenizers.normalizers import NFKD
174        >>> normalizer = NFKD()
175        >>> normalizer.normalize_str("fine")
176        'fine'
177    """
178    def __new__(cls, /) -> NFKD: ...
179
180@final
181class Nmt(Normalizer):
182    """
183    Nmt normalizer
184
185    Normalizer used in the Google NMT pipeline. It handles various text cleaning tasks
186    including removing control characters, normalizing whitespace, and replacing certain
187    Unicode characters. This is equivalent to the normalization done in the original
188    SentencePiece NMT preprocessing.
189
190    Example::
191
192        >>> from tokenizers.normalizers import Nmt
193        >>> normalizer = Nmt()
194        >>> normalizer.normalize_str("Hello\x00World")
195        'Hello World'
196    """
197    def __new__(cls, /) -> Nmt: ...
198
199class Normalizer:
200    """
201    Base class for all normalizers
202
203    This class is not supposed to be instantiated directly. Instead, any implementation of a
204    Normalizer will return an instance of this class when instantiated.
205    """
206    def __getstate__(self, /) -> Any: ...
207    def __repr__(self, /) -> str: ...
208    def __setstate__(self, /, state: Any) -> None: ...
209    def __str__(self, /) -> str: ...
210    @staticmethod
211    def custom(obj: Any) -> Normalizer: ...
212    def normalize(self, /, normalized: NormalizedString | Any) -> None:
213        """
214        Normalize a :class:`~tokenizers.NormalizedString` in-place
215
216        This method allows to modify a :class:`~tokenizers.NormalizedString` to
217        keep track of the alignment information. If you just want to see the result
218        of the normalization on a raw string, you can use
219        :meth:`~tokenizers.normalizers.Normalizer.normalize_str`
220
221        Args:
222            normalized (:class:`~tokenizers.NormalizedString`):
223                The normalized string on which to apply this
224                :class:`~tokenizers.normalizers.Normalizer`
225        """
226    def normalize_str(self, /, sequence: str) -> str:
227        """
228        Normalize the given string
229
230        This method provides a way to visualize the effect of a
231        :class:`~tokenizers.normalizers.Normalizer` but it does not keep track of the alignment
232        information. If you need to get/convert offsets, you can use
233        :meth:`~tokenizers.normalizers.Normalizer.normalize`
234
235        Args:
236            sequence (:obj:`str`):
237                A string to normalize
238
239        Returns:
240            :obj:`str`: A string after normalization
241        """
242
243@final
244class Precompiled(Normalizer):
245    """
246    Precompiled normalizer
247
248    A normalizer that uses a precompiled character map built from a SentencePiece model.
249    This normalizer is automatically extracted from SentencePiece ``.model`` files and
250    should not be constructed manually — it is used internally for full compatibility
251    with SentencePiece-based tokenizers.
252
253    Args:
254        precompiled_charsmap (:obj:`bytes`):
255            The raw bytes of the precompiled character map, as found inside a
256            SentencePiece ``.model`` file.
257    """
258    def __new__(cls, /, precompiled_charsmap: Sequence2[int]) -> Precompiled: ...
259
260@final
261class Prepend(Normalizer):
262    """
263    Prepend normalizer
264
265    Prepends a given string to the beginning of the input. This is typically used to
266    add a meta-symbol such as ``▁`` (U+2581) at the start of each sequence, which is
267    the convention used by SentencePiece-based models to indicate that a token appears
268    at the start of a word.
269
270    Args:
271        prepend (:obj:`str`, defaults to :obj:`"▁"`):
272            The string to prepend to the input.
273
274    Example::
275
276        >>> from tokenizers.normalizers import Prepend
277        >>> normalizer = Prepend("▁")
278        >>> normalizer.normalize_str("hello")
279        '▁hello'
280    """
281    def __new__(cls, /, prepend: str = ...) -> Prepend: ...
282    @property
283    def prepend(self, /) -> str: ...
284    @prepend.setter
285    def prepend(self, /, prepend: str) -> None: ...
286
287@final
288class Replace(Normalizer):
289    """
290    Replace normalizer
291
292    Replaces occurrences of a pattern in the input string with the given content.
293    The pattern can be either a plain string or a regular expression wrapped in
294    :class:`~tokenizers.Regex`.
295
296    Args:
297        pattern (:obj:`str` or :class:`~tokenizers.Regex`):
298            The pattern to search for. Use a plain string for literal replacement,
299            or wrap a regex pattern in :class:`~tokenizers.Regex` for regex replacement.
300
301        content (:obj:`str`):
302            The string to replace each match with.
303
304    Example::
305
306        >>> from tokenizers import Regex
307        >>> from tokenizers.normalizers import Replace
308        >>> # Replace a literal string
309        >>> Replace(".", " ").normalize_str("hello.world")
310        'hello world'
311        >>> # Replace using a regex
312        >>> Replace(Regex(r"\s+"), " ").normalize_str("hello   world")
313        'hello world'
314    """
315    def __new__(cls, /, pattern: str | Regex, content: str) -> Replace: ...
316    @property
317    def content(self, /) -> str: ...
318    @content.setter
319    def content(self, /, content: str) -> None: ...
320    @property
321    def pattern(self, /) -> None: ...
322    @pattern.setter
323    def pattern(self, /, _pattern: str | Regex) -> None: ...
324
325@final
326class Sequence(Normalizer):
327    """
328    Allows concatenating multiple other Normalizer as a Sequence.
329    All the normalizers run in sequence in the given order
330
331    Args:
332        normalizers (:obj:`List[Normalizer]`):
333            A list of Normalizer to be run as a sequence
334
335    Example::
336
337        >>> from tokenizers.normalizers import NFD, Lowercase, StripAccents, Sequence
338        >>> normalizer = Sequence([NFD(), Lowercase(), StripAccents()])
339        >>> normalizer.normalize_str("Héllo Wörld")
340        'hello world'
341    """
342    def __getitem__(self, /, index: int) -> Any: ...
343    def __getnewargs__(self, /) -> tuple: ...
344    def __len__(self, /) -> int: ...
345    def __new__(cls, /, normalizers: list) -> Sequence: ...
346    def __setitem__(self, /, index: int, value: Any) -> None: ...
347
348@final
349class Strip(Normalizer):
350    """
351    Strip normalizer
352
353    Removes leading and/or trailing whitespace from the input string.
354
355    Args:
356        left (:obj:`bool`, defaults to :obj:`True`):
357            Whether to strip leading (left) whitespace.
358
359        right (:obj:`bool`, defaults to :obj:`True`):
360            Whether to strip trailing (right) whitespace.
361
362    Example::
363
364        >>> from tokenizers.normalizers import Strip
365        >>> normalizer = Strip()
366        >>> normalizer.normalize_str("  hello world  ")
367        'hello world'
368        >>> Strip(right=False).normalize_str("  hello  ")
369        'hello  '
370    """
371    def __new__(cls, /, left: bool = True, right: bool = True) -> Strip: ...
372    @property
373    def left(self, /) -> bool: ...
374    @left.setter
375    def left(self, /, left: bool) -> None: ...
376    @property
377    def right(self, /) -> bool: ...
378    @right.setter
379    def right(self, /, right: bool) -> None: ...
380
381@final
382class StripAccents(Normalizer):
383    """
384    StripAccents normalizer
385
386    Strips all accent marks (combining diacritical characters) from the input. This
387    normalizer should typically be used after applying :class:`~tokenizers.normalizers.NFD`
388    or :class:`~tokenizers.normalizers.NFKD` decomposition, which separates base
389    characters from their combining accents.
390
391    Example::
392
393        >>> from tokenizers.normalizers import NFD, StripAccents, Sequence
394        >>> normalizer = Sequence([NFD(), StripAccents()])
395        >>> normalizer.normalize_str("café")
396        'cafe'
397    """
398    def __new__(cls, /) -> StripAccents: ...
399 
codekingpro/portable-devtools · Team Ai