eduagarcia/multilingual_tokenizer_benchmark
Multilingual Tokenizer Benchmark More details of each subset like word count, character count, original sources, etc, can be found in the dataset_meta.yaml file in the repository root. Natural language word count functions Download spacy models pip install ntlk spacy pygments underthesea camel-tools python -m spacy download ko_core_news_sm python -m spacy download ja_core_news_sm python -m spacy download zh_core_web_sm import nltk nltk.download('punkt_tab')… See the full description on the dataset page: https://huggingface.co/datasets/eduagarcia/multilingual_tokenizer_benchmark.
Multilingual Tokenizer Benchmark
More details of each subset like word count, character count, original sources, etc, can be found in the dataset_meta.yaml file in the repository root.
Natural language word count functions
Download spacy models
pip install ntlk spacy pygments underthesea camel-tools
python -m spacy download ko_core_news_sm
python -m spacy download ja_core_news_sm
python -m spacy download zh_core_web_smimport nltk
nltk.download('punkt_tab')
nltk.download('words')
from nltk import word_tokenize as word_tokenize_nltk
from underthesea import word_tokenize as word_tokenize_underthesea
from camel_tools.tokenizers.word import simple_word_tokenize as word_tokenize_camel_tools
import spacy
langs_nltk = {
'en': 'english',
'pt': 'portuguese',
'it': 'italian',
'pl': 'polish',
'de': 'german',
'es': 'spanish',
'fr': 'french',
'ru': 'russian',
}
langs_spacy = {
'ja': spacy.load('ja_core_news_sm', disable=['parser', 'ner', 'lemmatizer', 'tagger', 'attribute_ruler']),
'zh': spacy.load('zh_core_web_sm', disable=['parser', 'ner', 'lemmatizer', 'tagger', 'attribute_ruler']),
'ko': spacy.load('ko_core_news_sm', disable=['parser', 'ner', 'lemmatizer', 'tagger', 'attribute_ruler']),
}
def word_tokenize_spacy(text, model):
doc = model(text)
return [token.text for token in doc if not token.is_space]
def word_tokenize(text, language):
if language in langs_spacy:
return word_tokenize_spacy(text, langs_spacy[language])
elif language in langs_nltk:
return word_tokenize_nltk(text, langs_nltk[language])
elif language == 'vi':
return word_tokenize_underthesea(text)
elif language == 'ar':
return word_tokenize_camel_tools(text)
else:
raise ValueError(f"Language {language} not supported")
#### Usage Example ####
word_count_pt = len(word_tokenize("Olá Mundo.", 'pt'))
word_count_ko = len(word_tokenize("이것은 한국어 문장입니다.", 'ko'))Code word count function
import nltk
nltk.download('punkt_tab')
from nltk import word_tokenize
from pygments import lex
from pygments.lexers import get_lexer_by_name
from pygments.token import Token
NATURAL_LANGUAGE_PARENT_TOKEN_TYPES = {
Token.Comment,
Token.Literal.String,
Token.Text,
Token.Generic.Heading, # For Markdown,
Token.Generic.Subheading, # For Markdown,
Token.Generic.Emph, # For Markdown *italic*
Token.Generic.Strong, # For Markdown **bold**
Token.Generic.Output, # For shell output,
Token.Generic.Error, # Error messages
Token.Generic.Traceback, # Traceback messages
}
def _is_natural_language_like(tok_type) -> bool:
for nl_parent_type in NATURAL_LANGUAGE_PARENT_TOKEN_TYPES:
if tok_type in nl_parent_type: # This checks if tok_type is nl_parent_type OR a child
return True
return False
def pygments_tokenize(code: str, language: str) -> list[str]:
lexer = get_lexer_by_name(language, stripnl=True, stripall=True)
tokens = lex(code, lexer)
result = []
for tok_type, value in tokens:
if _is_natural_language_like(tok_type):
# Recursively tokenize comment or string content. Use the default english tokenizer.
result.extend(word_tokenize(value))
else:
# Tokenize symbol/keyword/identifier at surface level
result.append(value)
return [t for t in result if t.strip() != '']
#### Usage Example ####
word_count_json = len(pygments_tokenize('{"hello": 1}', 'json'))
word_count_python = len(pygments_tokenize("lambda x: print('hello')", "python"))