Team Ai
Datasetpublic

codekingpro/portable-devtools

sourceHugging Faceupdated 5mo agoView on Hugging Face
1likes15kdownloads
tokenizer.py121 linesDownload Raw Back to utils
1# This code is a modified copy of the `NLTKWordTokenizer` class from `NLTK` library.2 3import re4 5 6class SimpleTokenizer:7    @staticmethod8    def tokenize(text: str) -> list[str]:9        text = re.sub(r"[^\w]", " ", text.lower())10        text = re.sub(r"\s+", " ", text)11 12        return text.strip().split()13 14 15class WordTokenizer:16    """The tokenizer is "destructive" such that the regexes applied will munge the17    input string to a state beyond re-construction.18    """19 20    # Starting quotes.21    STARTING_QUOTES = [22        (re.compile("([«“‘„]|[`]+)", re.U), r" \1 "),23        (re.compile(r"^\""), r"``"),24        (re.compile(r"(``)"), r" \1 "),25        (re.compile(r"([ \(\[{<])(\"|\'{2})"), r"\1 `` "),26        (re.compile(r"(?i)(\')(?!re|ve|ll|m|t|s|d|n)(\w)\b", re.U), r"\1 \2"),27    ]28 29    # Ending quotes.30    ENDING_QUOTES = [31        (re.compile("([»”’])", re.U), r" \1 "),32        (re.compile(r"''"), " '' "),33        (re.compile(r'"'), " '' "),34        (re.compile(r"([^' ])('[sS]|'[mM]|'[dD]|') "), r"\1 \2 "),35        (re.compile(r"([^' ])('ll|'LL|'re|'RE|'ve|'VE|n't|N'T) "), r"\1 \2 "),36    ]37 38    # Punctuation.39    PUNCTUATION = [40        (re.compile(r'([^\.])(\.)([\]\)}>"\'' "»”’ " r"]*)\s*$", re.U), r"\1 \2 \3 "),41        (re.compile(r"([:,])([^\d])"), r" \1 \2"),42        (re.compile(r"([:,])$"), r" \1 "),43        (44            re.compile(r"\.{2,}", re.U),45            r" \g<0> ",46        ),47        (re.compile(r"[;@#$%&]"), r" \g<0> "),48        (49            re.compile(r'([^\.])(\.)([\]\)}>"\']*)\s*$'),50            r"\1 \2\3 ",51        ),  # Handles the final period.52        (re.compile(r"[?!]"), r" \g<0> "),53        (re.compile(r"([^'])' "), r"\1 ' "),54        (55            re.compile(r"[*]", re.U),56            r" \g<0> ",57        ),58    ]59 60    # Pads parentheses61    PARENS_BRACKETS = (re.compile(r"[\]\[\(\)\{\}\<\>]"), r" \g<0> ")62    DOUBLE_DASHES = (re.compile(r"--"), r" -- ")63 64    # List of contractions adapted from Robert MacIntyre's tokenizer.65    CONTRACTIONS2 = [66        re.compile(pattern)67        for pattern in (68            r"(?i)\b(can)(?#X)(not)\b",69            r"(?i)\b(d)(?#X)('ye)\b",70            r"(?i)\b(gim)(?#X)(me)\b",71            r"(?i)\b(gon)(?#X)(na)\b",72            r"(?i)\b(got)(?#X)(ta)\b",73            r"(?i)\b(lem)(?#X)(me)\b",74            r"(?i)\b(more)(?#X)('n)\b",75            r"(?i)\b(wan)(?#X)(na)(?=\s)",76        )77    ]78    CONTRACTIONS3 = [79        re.compile(pattern) for pattern in (r"(?i) ('t)(?#X)(is)\b", r"(?i) ('t)(?#X)(was)\b")80    ]81 82    @classmethod83    def tokenize(cls, text: str) -> list[str]:84        """Return a tokenized copy of `text`.85 86        >>> s = '''Good muffins cost $3.88 (roughly 3,36 euros)\nin New York.'''87        >>> WordTokenizer().tokenize(s)88        ['Good', 'muffins', 'cost', '$', '3.88', '(', 'roughly', '3,36', 'euros', ')', 'in', 'New', 'York', '.']89 90        Args:91            text: The text to be tokenized.92 93        Returns:94            A list of tokens.95        """96        for regexp, substitution in cls.STARTING_QUOTES:97            text = regexp.sub(substitution, text)98 99        for regexp, substitution in cls.PUNCTUATION:100            text = regexp.sub(substitution, text)101 102        # Handles parentheses.103        regexp, substitution = cls.PARENS_BRACKETS104        text = regexp.sub(substitution, text)105 106        # Handles double dash.107        regexp, substitution = cls.DOUBLE_DASHES108        text = regexp.sub(substitution, text)109 110        # add extra space to make things easier111        text = " " + text + " "112 113        for regexp, substitution in cls.ENDING_QUOTES:114            text = regexp.sub(substitution, text)115 116        for regexp in cls.CONTRACTIONS2:117            text = regexp.sub(r" \1 \2 ", text)118        for regexp in cls.CONTRACTIONS3:119            text = regexp.sub(r" \1 \2 ", text)120        return text.split()121 
codekingpro/portable-devtools · Team Ai