codekingpro/portable-devtools
114k
1# Generated content DO NOT EDIT
2class Trainer:
3 """
4 Base class for all trainers
5
6 This class is not supposed to be instantiated directly. Instead, any implementation of a
7 Trainer will return an instance of this class when instantiated.
8 """
9 def __getstate__(self):
10 """ """
11 pass
12
13 def __setstate__(self, state):
14 """ """
15 pass
16
17class BpeTrainer(Trainer):
18 """
19 Trainer capable of training a BPE model
20
21 Args:
22 vocab_size (:obj:`int`, `optional`):
23 The size of the final vocabulary, including all tokens and alphabet.
24
25 min_frequency (:obj:`int`, `optional`):
26 The minimum frequency a pair should have in order to be merged.
27
28 show_progress (:obj:`bool`, `optional`):
29 Whether to show progress bars while training.
30
31 special_tokens (:obj:`List[Union[str, AddedToken]]`, `optional`):
32 A list of special tokens the model should know of.
33
34 limit_alphabet (:obj:`int`, `optional`):
35 The maximum different characters to keep in the alphabet.
36
37 initial_alphabet (:obj:`List[str]`, `optional`):
38 A list of characters to include in the initial alphabet, even
39 if not seen in the training dataset.
40 If the strings contain more than one character, only the first one
41 is kept.
42
43 continuing_subword_prefix (:obj:`str`, `optional`):
44 A prefix to be used for every subword that is not a beginning-of-word.
45
46 end_of_word_suffix (:obj:`str`, `optional`):
47 A suffix to be used for every subword that is a end-of-word.
48
49 max_token_length (:obj:`int`, `optional`):
50 Prevents creating tokens longer than the specified size.
51 This can help with reducing polluting your vocabulary with
52 highly repetitive tokens like `======` for wikipedia
53
54 """
55 def __init__(
56 self,
57 vocab_size=30000,
58 min_frequency=0,
59 show_progress=True,
60 special_tokens=[],
61 limit_alphabet=None,
62 initial_alphabet=[],
63 continuing_subword_prefix=None,
64 end_of_word_suffix=None,
65 max_token_length=None,
66 words={},
67 ):
68 pass
69
70 def __getstate__(self):
71 """ """
72 pass
73
74 def __setstate__(self, state):
75 """ """
76 pass
77
78 @property
79 def continuing_subword_prefix(self):
80 """ """
81 pass
82
83 @continuing_subword_prefix.setter
84 def continuing_subword_prefix(self, value):
85 """ """
86 pass
87
88 @property
89 def end_of_word_suffix(self):
90 """ """
91 pass
92
93 @end_of_word_suffix.setter
94 def end_of_word_suffix(self, value):
95 """ """
96 pass
97
98 @property
99 def initial_alphabet(self):
100 """ """
101 pass
102
103 @initial_alphabet.setter
104 def initial_alphabet(self, value):
105 """ """
106 pass
107
108 @property
109 def limit_alphabet(self):
110 """ """
111 pass
112
113 @limit_alphabet.setter
114 def limit_alphabet(self, value):
115 """ """
116 pass
117
118 @property
119 def max_token_length(self):
120 """ """
121 pass
122
123 @max_token_length.setter
124 def max_token_length(self, value):
125 """ """
126 pass
127
128 @property
129 def min_frequency(self):
130 """ """
131 pass
132
133 @min_frequency.setter
134 def min_frequency(self, value):
135 """ """
136 pass
137
138 @property
139 def show_progress(self):
140 """ """
141 pass
142
143 @show_progress.setter
144 def show_progress(self, value):
145 """ """
146 pass
147
148 @property
149 def special_tokens(self):
150 """ """
151 pass
152
153 @special_tokens.setter
154 def special_tokens(self, value):
155 """ """
156 pass
157
158 @property
159 def vocab_size(self):
160 """ """
161 pass
162
163 @vocab_size.setter
164 def vocab_size(self, value):
165 """ """
166 pass
167
168class UnigramTrainer(Trainer):
169 """
170 Trainer capable of training a Unigram model
171
172 Args:
173 vocab_size (:obj:`int`):
174 The size of the final vocabulary, including all tokens and alphabet.
175
176 show_progress (:obj:`bool`):
177 Whether to show progress bars while training.
178
179 special_tokens (:obj:`List[Union[str, AddedToken]]`):
180 A list of special tokens the model should know of.
181
182 initial_alphabet (:obj:`List[str]`):
183 A list of characters to include in the initial alphabet, even
184 if not seen in the training dataset.
185 If the strings contain more than one character, only the first one
186 is kept.
187
188 shrinking_factor (:obj:`float`):
189 The shrinking factor used at each step of the training to prune the
190 vocabulary.
191
192 unk_token (:obj:`str`):
193 The token used for out-of-vocabulary tokens.
194
195 max_piece_length (:obj:`int`):
196 The maximum length of a given token.
197
198 n_sub_iterations (:obj:`int`):
199 The number of iterations of the EM algorithm to perform before
200 pruning the vocabulary.
201 """
202 def __init__(
203 self,
204 vocab_size=8000,
205 show_progress=True,
206 special_tokens=[],
207 initial_alphabet=[],
208 shrinking_factor=0.75,
209 unk_token=None,
210 max_piece_length=16,
211 n_sub_iterations=2,
212 ):
213 pass
214
215 def __getstate__(self):
216 """ """
217 pass
218
219 def __setstate__(self, state):
220 """ """
221 pass
222
223 @property
224 def initial_alphabet(self):
225 """ """
226 pass
227
228 @initial_alphabet.setter
229 def initial_alphabet(self, value):
230 """ """
231 pass
232
233 @property
234 def show_progress(self):
235 """ """
236 pass
237
238 @show_progress.setter
239 def show_progress(self, value):
240 """ """
241 pass
242
243 @property
244 def special_tokens(self):
245 """ """
246 pass
247
248 @special_tokens.setter
249 def special_tokens(self, value):
250 """ """
251 pass
252
253 @property
254 def vocab_size(self):
255 """ """
256 pass
257
258 @vocab_size.setter
259 def vocab_size(self, value):
260 """ """
261 pass
262
263class WordLevelTrainer(Trainer):
264 """
265 Trainer capable of training a WorldLevel model
266
267 Args:
268 vocab_size (:obj:`int`, `optional`):
269 The size of the final vocabulary, including all tokens and alphabet.
270
271 min_frequency (:obj:`int`, `optional`):
272 The minimum frequency a pair should have in order to be merged.
273
274 show_progress (:obj:`bool`, `optional`):
275 Whether to show progress bars while training.
276
277 special_tokens (:obj:`List[Union[str, AddedToken]]`):
278 A list of special tokens the model should know of.
279 """
280 def __init__(self, vocab_size=30000, min_frequency=0, show_progress=True, special_tokens=[]):
281 pass
282
283 def __getstate__(self):
284 """ """
285 pass
286
287 def __setstate__(self, state):
288 """ """
289 pass
290
291 @property
292 def min_frequency(self):
293 """ """
294 pass
295
296 @min_frequency.setter
297 def min_frequency(self, value):
298 """ """
299 pass
300
301 @property
302 def show_progress(self):
303 """ """
304 pass
305
306 @show_progress.setter
307 def show_progress(self, value):
308 """ """
309 pass
310
311 @property
312 def special_tokens(self):
313 """ """
314 pass
315
316 @special_tokens.setter
317 def special_tokens(self, value):
318 """ """
319 pass
320
321 @property
322 def vocab_size(self):
323 """ """
324 pass
325
326 @vocab_size.setter
327 def vocab_size(self, value):
328 """ """
329 pass
330
331class WordPieceTrainer(Trainer):
332 """
333 Trainer capable of training a WordPiece model
334
335 Args:
336 vocab_size (:obj:`int`, `optional`):
337 The size of the final vocabulary, including all tokens and alphabet.
338
339 min_frequency (:obj:`int`, `optional`):
340 The minimum frequency a pair should have in order to be merged.
341
342 show_progress (:obj:`bool`, `optional`):
343 Whether to show progress bars while training.
344
345 special_tokens (:obj:`List[Union[str, AddedToken]]`, `optional`):
346 A list of special tokens the model should know of.
347
348 limit_alphabet (:obj:`int`, `optional`):
349 The maximum different characters to keep in the alphabet.
350
351 initial_alphabet (:obj:`List[str]`, `optional`):
352 A list of characters to include in the initial alphabet, even
353 if not seen in the training dataset.
354 If the strings contain more than one character, only the first one
355 is kept.
356
357 continuing_subword_prefix (:obj:`str`, `optional`):
358 A prefix to be used for every subword that is not a beginning-of-word.
359
360 end_of_word_suffix (:obj:`str`, `optional`):
361 A suffix to be used for every subword that is a end-of-word.
362 """
363 def __init__(
364 self,
365 vocab_size=30000,
366 min_frequency=0,
367 show_progress=True,
368 special_tokens=[],
369 limit_alphabet=None,
370 initial_alphabet=[],
371 continuing_subword_prefix="##",
372 end_of_word_suffix=None,
373 ):
374 pass
375
376 def __getstate__(self):
377 """ """
378 pass
379
380 def __setstate__(self, state):
381 """ """
382 pass
383
384 @property
385 def continuing_subword_prefix(self):
386 """ """
387 pass
388
389 @continuing_subword_prefix.setter
390 def continuing_subword_prefix(self, value):
391 """ """
392 pass
393
394 @property
395 def end_of_word_suffix(self):
396 """ """
397 pass
398
399 @end_of_word_suffix.setter
400 def end_of_word_suffix(self, value):
401 """ """
402 pass
403
404 @property
405 def initial_alphabet(self):
406 """ """
407 pass
408
409 @initial_alphabet.setter
410 def initial_alphabet(self, value):
411 """ """
412 pass
413
414 @property
415 def limit_alphabet(self):
416 """ """
417 pass
418
419 @limit_alphabet.setter
420 def limit_alphabet(self, value):
421 """ """
422 pass
423
424 @property
425 def min_frequency(self):
426 """ """
427 pass
428
429 @min_frequency.setter
430 def min_frequency(self, value):
431 """ """
432 pass
433
434 @property
435 def show_progress(self):
436 """ """
437 pass
438
439 @show_progress.setter
440 def show_progress(self, value):
441 """ """
442 pass
443
444 @property
445 def special_tokens(self):
446 """ """
447 pass
448
449 @special_tokens.setter
450 def special_tokens(self, value):
451 """ """
452 pass
453
454 @property
455 def vocab_size(self):
456 """ """
457 pass
458
459 @vocab_size.setter
460 def vocab_size(self, value):
461 """ """
462 pass
463 