SinaAhmadi/ScriptNormalization
4
1from pathlib import Path2from functools import partial3 4from joeynmt.prediction import predict5from joeynmt.helpers import (6 check_version,7 load_checkpoint,8 load_config,9 parse_train_args,10 resolve_ckpt_path,11 12)13from joeynmt.model import build_model14from joeynmt.tokenizers import build_tokenizer15from joeynmt.vocabulary import build_vocab16from joeynmt.datasets import build_dataset17 18import gradio as gr19 20languages_scripts = {21 "Azeri Turkish in Persian": "AzeriTurkish-Persian",22 "Central Kurdish in Arabic": "Sorani-Arabic",23 "Central Kurdish in Persian": "Sorani-Persian",24 "Gilaki in Persian": "Gilaki-Persian",25 "Gorani in Arabic": "Gorani-Arabic",26 "Gorani in Central Kurdish": "Gorani-Sorani",27 "Gorani in Persian": "Gorani-Persian",28 "Kashmiri in Urdu": "Kashmiri-Urdu",29 "Mazandarani in Persian": "Mazandarani-Persian",30 "Northern Kurdish in Arabic": "Kurmanji-Arabic",31 "Northern Kurdish in Persian": "Kurmanji-Persian",32 "Sindhi in Urdu": "Sindhi-Urdu"33}34 35def normalize(text, language_script):36 37 cfg_file = "./models/%s/config.yaml"%languages_scripts[language_script]38 ckpt = "./models/%s/best.ckpt"%languages_scripts[language_script]39 40 cfg = load_config(Path(cfg_file))41 # parse and validate cfg42 model_dir, load_model, device, n_gpu, num_workers, _, fp16 = parse_train_args(43 cfg["training"], mode="prediction")44 test_cfg = cfg["testing"]45 src_cfg = cfg["data"]["src"]46 trg_cfg = cfg["data"]["trg"]47 48 load_model = load_model if ckpt is None else Path(ckpt)49 ckpt = resolve_ckpt_path(load_model, model_dir)50 51 src_vocab, trg_vocab = build_vocab(cfg["data"], model_dir=model_dir)52 53 model = build_model(cfg["model"], src_vocab=src_vocab, trg_vocab=trg_vocab)54 55 # load model state from disk56 model_checkpoint = load_checkpoint(ckpt, device=device)57 model.load_state_dict(model_checkpoint["model_state"])58 59 if device.type == "cuda":60 model.to(device)61 62 tokenizer = build_tokenizer(cfg["data"])63 sequence_encoder = {64 src_cfg["lang"]: partial(src_vocab.sentences_to_ids, bos=False, eos=True),65 trg_cfg["lang"]: None,66 }67 68 test_cfg["batch_size"] = 1 # CAUTION: this will raise an error if n_gpus > 169 test_cfg["batch_type"] = "sentence"70 71 test_data = build_dataset(72 dataset_type="stream",73 path=None,74 src_lang=src_cfg["lang"],75 trg_lang=trg_cfg["lang"],76 split="test",77 tokenizer=tokenizer,78 sequence_encoder=sequence_encoder,79 )80 test_data.set_item(text.strip())81 82 cfg=test_cfg83 _, _, hypotheses, trg_tokens, trg_scores, _ = predict(84 model=model,85 data=test_data,86 compute_loss=False,87 device=device,88 n_gpu=n_gpu,89 normalization="none",90 num_workers=num_workers,91 cfg=cfg,92 fp16=fp16,93 )94 return hypotheses[0]95 96title = """97<center><strong><font size='8'>Script Normalization for Unconventional Writing<font></strong></center>98 99<div align="center">100 <img src="https://raw.githubusercontent.com/sinaahmadi/ScriptNormalization/main/Perso-Arabic_scripts.jpg" alt="Perso-Arabic scripts used by the target languages in our paper" width="400">101</div>102 103<h3 style="font-weight: 450; font-size: 1rem; margin: 0rem"> 104 [<a href="https://sinaahmadi.github.io/docs/articles/ahmadi2023acl.pdf" style="color:blue;">Paper (ACL 2023)</a>] 105 [<a href="https://sinaahmadi.github.io/docs/slides/ahmadi2023acl_slides.pdf" style="color:blue;">Slides</a>]106 [<a href="https://github.com/sinaahmadi/ScriptNormalization" style="color:blue;">GitHub</a>]107 [<a href="https://s3.amazonaws.com/pf-user-files-01/u-59356/uploads/2023-06-04/rw32pwp/ACL2023.mp4" style="color:blue;">Presentation</a>]108</h3>109 """110 111description = """112<ul>113 <li style="font-size:120%;">"<em>mar7aba!</em>"</li>114 <li style="font-size:120%;">"<em>هاو ئار یوو؟</em>"</li>115 <li style="font-size:120%;">"<em>Μπιάνβενου α σετ ντεμό!</em>"</li>116</ul>117 118<p style="font-size:120%;">What do all these sentences have in common? Being greeted in Arabic with "<em>mar7aba</em>" written in the Latin script, then asked how you are ("<em>هاو ئار یوو؟</em>") in English using the Perso-Arabic script of Kurdish and then, welcomed to this demo in French ("<em>Μπιάνβενου α σετ ντεμό!</em>") written in Greek script. All these sentences are written in an <strong>unconventional</strong> script.</p>119 120<p style="font-size:120%;">Although you may find these sentences risible, unconventional writing is a common practice among millions of speakers in bilingual communities. In our paper entitled "<a href="https://sinaahmadi.github.io/docs/articles/ahmadi2023acl.pdf" target="_blank"><strong>Script Normalization for Unconventional Writing of Under-Resourced Languages in Bilingual Communities</strong></a>", we shed light on this problem and propose an approach to normalize noisy text written in unconventional writing.</p>121 122<p style="font-size:120%;">This demo deploys a few models that are trained for <strong>the normalization of unconventional writing</strong>. Please note that this tool is not a spell-checker and cannot correct errors beyond character normalization. For better performance, you can apply hard-coded rules on the input and then pass it to the models, hence a hybrid system.</p>123 124<p style="font-size:120%;">For more information, you can check out the project on GitHub too: <a href="https://github.com/sinaahmadi/ScriptNormalization" target="_blank"><strong>https://github.com/sinaahmadi/ScriptNormalization</strong></a></p>125"""126 127examples = [128 ["بو شهرین نوفوسو ، 2014 نجی ایلين نوفوس ساییمی اساسيندا 41 نفر ایمیش .", "Azeri Turkish in Persian"],#"بۇ شهرین نۆفوسو ، 2014 نجی ایلين نۆفوس ساییمی اساسيندا 41 نفر ایمیش ."129 ["ياخوا تةمةن دريژبيت بوئةم ميللةتة", "Central Kurdish in Arabic"],130 ["یکیک له جوانیکانی ام شاره جوانه", "Central Kurdish in Persian"],131 ["نمک درهٰ مردوم گيلک ايسن ؤ اوشان زوان ني گيلکي ايسه .", "Gilaki in Persian"],132 ["شؤنةو اانةيةرة گةشت و گلي ناجارانةو اؤجالاني دةستش پنةكةرد", "Gorani in Arabic"], #شۆنەو ئانەیەرە گەشت و گێڵی ناچارانەو ئۆجالانی دەستش پنەکەرد133 ["ڕوٙو زوانی ئەذایی چەنی پەیذابی ؟", "Gorani in Central Kurdish"], # ڕوٙو زوانی ئەڎایی چەنی پەیڎابی ؟134 ["هنگامکان ظميٛ ر چمان ، بپا کريٛلي بيشان :", "Gorani in Persian"], # هەنگامەکان وزمیٛ وەرو چەمان ، بەپاو کریٛڵی بیەشان :135 ["ربعی بن افکل اُسے اَکھ صُحابی .", "Kashmiri in Urdu"], # ربعی بن افکل ٲسؠ اَکھ صُحابی .136 ["اینتا زون گنشکرون 85 میلیون نفر هسن", "Mazandarani in Persian"], # اینتا زوون گِنِشکَرون 85 میلیون نفر هسنه137 ["بة رطكا هة صطئن ژ دل هاطة بة لافكرن", "Northern Kurdish in Arabic"], #پەرتوکا هەستێن ژ دل هاتە بەلافکرن138 ["ثرکى همرنگ نرميني دويت هندک قوناغين دي ببريت", "Northern Kurdish in Persian"], # سەرەکی هەمەرەنگ نەرمینێ دڤێت هندەک قوناغێن دی ببڕیت139 ["ہتی کجھ اپ ۽ تمام دائون ترینون بیھندیون آھن .", "Sindhi in Urdu"] # هتي ڪجھ اپ ۽ تمام ڊائون ٽرينون بيھنديون آھن .140]141 142 143article = """144<div style="text-align: justify; max-width: 1200px; margin: 20px auto;">145 <h3 style="font-weight: 450; font-size: 1rem; margin: 0rem">146 <b>Created and deployed by Sina Ahmadi <a href="https://sinaahmadi.github.io/">(https://sinaahmadi.github.io/)</a>.147 </h3>148</div>149 """150demo = gr.Interface(151 title=title,152 description=description,153 fn=normalize,154 inputs=[155 gr.Textbox(lines=4, label="Noisy Text \U0001F974"),156 gr.Dropdown(label="Language in unconventional script", choices=sorted(list(languages_scripts.keys()))),157 ],158 outputs=gr.Textbox(label="Normalized Text \U0001F642"),159 examples=examples,160 article=article,161 examples_per_page=20,162 cache_examples=False 163)164 165demo.launch()166 