codekingpro/portable-devtools
115k
1# This code is a modified copy of the `NLTKWordTokenizer` class from `NLTK` library.2 3import re4 5 6class SimpleTokenizer:7 @staticmethod8 def tokenize(text: str) -> list[str]:9 text = re.sub(r"[^\w]", " ", text.lower())10 text = re.sub(r"\s+", " ", text)11 12 return text.strip().split()13 14 15class WordTokenizer:16 """The tokenizer is "destructive" such that the regexes applied will munge the17 input string to a state beyond re-construction.18 """19 20 # Starting quotes.21 STARTING_QUOTES = [22 (re.compile("([«“‘„]|[`]+)", re.U), r" \1 "),23 (re.compile(r"^\""), r"``"),24 (re.compile(r"(``)"), r" \1 "),25 (re.compile(r"([ \(\[{<])(\"|\'{2})"), r"\1 `` "),26 (re.compile(r"(?i)(\')(?!re|ve|ll|m|t|s|d|n)(\w)\b", re.U), r"\1 \2"),27 ]28 29 # Ending quotes.30 ENDING_QUOTES = [31 (re.compile("([»”’])", re.U), r" \1 "),32 (re.compile(r"''"), " '' "),33 (re.compile(r'"'), " '' "),34 (re.compile(r"([^' ])('[sS]|'[mM]|'[dD]|') "), r"\1 \2 "),35 (re.compile(r"([^' ])('ll|'LL|'re|'RE|'ve|'VE|n't|N'T) "), r"\1 \2 "),36 ]37 38 # Punctuation.39 PUNCTUATION = [40 (re.compile(r'([^\.])(\.)([\]\)}>"\'' "»”’ " r"]*)\s*$", re.U), r"\1 \2 \3 "),41 (re.compile(r"([:,])([^\d])"), r" \1 \2"),42 (re.compile(r"([:,])$"), r" \1 "),43 (44 re.compile(r"\.{2,}", re.U),45 r" \g<0> ",46 ),47 (re.compile(r"[;@#$%&]"), r" \g<0> "),48 (49 re.compile(r'([^\.])(\.)([\]\)}>"\']*)\s*$'),50 r"\1 \2\3 ",51 ), # Handles the final period.52 (re.compile(r"[?!]"), r" \g<0> "),53 (re.compile(r"([^'])' "), r"\1 ' "),54 (55 re.compile(r"[*]", re.U),56 r" \g<0> ",57 ),58 ]59 60 # Pads parentheses61 PARENS_BRACKETS = (re.compile(r"[\]\[\(\)\{\}\<\>]"), r" \g<0> ")62 DOUBLE_DASHES = (re.compile(r"--"), r" -- ")63 64 # List of contractions adapted from Robert MacIntyre's tokenizer.65 CONTRACTIONS2 = [66 re.compile(pattern)67 for pattern in (68 r"(?i)\b(can)(?#X)(not)\b",69 r"(?i)\b(d)(?#X)('ye)\b",70 r"(?i)\b(gim)(?#X)(me)\b",71 r"(?i)\b(gon)(?#X)(na)\b",72 r"(?i)\b(got)(?#X)(ta)\b",73 r"(?i)\b(lem)(?#X)(me)\b",74 r"(?i)\b(more)(?#X)('n)\b",75 r"(?i)\b(wan)(?#X)(na)(?=\s)",76 )77 ]78 CONTRACTIONS3 = [79 re.compile(pattern) for pattern in (r"(?i) ('t)(?#X)(is)\b", r"(?i) ('t)(?#X)(was)\b")80 ]81 82 @classmethod83 def tokenize(cls, text: str) -> list[str]:84 """Return a tokenized copy of `text`.85 86 >>> s = '''Good muffins cost $3.88 (roughly 3,36 euros)\nin New York.'''87 >>> WordTokenizer().tokenize(s)88 ['Good', 'muffins', 'cost', '$', '3.88', '(', 'roughly', '3,36', 'euros', ')', 'in', 'New', 'York', '.']89 90 Args:91 text: The text to be tokenized.92 93 Returns:94 A list of tokens.95 """96 for regexp, substitution in cls.STARTING_QUOTES:97 text = regexp.sub(substitution, text)98 99 for regexp, substitution in cls.PUNCTUATION:100 text = regexp.sub(substitution, text)101 102 # Handles parentheses.103 regexp, substitution = cls.PARENS_BRACKETS104 text = regexp.sub(substitution, text)105 106 # Handles double dash.107 regexp, substitution = cls.DOUBLE_DASHES108 text = regexp.sub(substitution, text)109 110 # add extra space to make things easier111 text = " " + text + " "112 113 for regexp, substitution in cls.ENDING_QUOTES:114 text = regexp.sub(substitution, text)115 116 for regexp in cls.CONTRACTIONS2:117 text = regexp.sub(r" \1 \2 ", text)118 for regexp in cls.CONTRACTIONS3:119 text = regexp.sub(r" \1 \2 ", text)120 return text.split()121 