Team Ai
Apppublic

SinaAhmadi/ScriptNormalization

sourceHugging Facemitupdated 2y agoView on Hugging Face
4likes
app.py166 linesDownload Raw Back to root
1from pathlib import Path2from functools import partial3 4from joeynmt.prediction import predict5from joeynmt.helpers import (6    check_version,7    load_checkpoint,8    load_config,9    parse_train_args,10    resolve_ckpt_path,11 12)13from joeynmt.model import build_model14from joeynmt.tokenizers import build_tokenizer15from joeynmt.vocabulary import build_vocab16from joeynmt.datasets import build_dataset17 18import gradio as gr19 20languages_scripts = {21    "Azeri Turkish in Persian": "AzeriTurkish-Persian",22    "Central Kurdish in Arabic": "Sorani-Arabic",23    "Central Kurdish in Persian": "Sorani-Persian",24    "Gilaki in Persian": "Gilaki-Persian",25    "Gorani in Arabic": "Gorani-Arabic",26    "Gorani in Central Kurdish": "Gorani-Sorani",27    "Gorani in Persian": "Gorani-Persian",28    "Kashmiri in Urdu": "Kashmiri-Urdu",29    "Mazandarani in Persian": "Mazandarani-Persian",30    "Northern Kurdish in Arabic": "Kurmanji-Arabic",31    "Northern Kurdish in Persian": "Kurmanji-Persian",32    "Sindhi in Urdu": "Sindhi-Urdu"33}34 35def normalize(text, language_script):36       37    cfg_file = "./models/%s/config.yaml"%languages_scripts[language_script]38    ckpt = "./models/%s/best.ckpt"%languages_scripts[language_script]39    40    cfg = load_config(Path(cfg_file))41        # parse and validate cfg42    model_dir, load_model, device, n_gpu, num_workers, _, fp16 = parse_train_args(43        cfg["training"], mode="prediction")44    test_cfg = cfg["testing"]45    src_cfg = cfg["data"]["src"]46    trg_cfg = cfg["data"]["trg"]47    48    load_model = load_model if ckpt is None else Path(ckpt)49    ckpt = resolve_ckpt_path(load_model, model_dir)50    51    src_vocab, trg_vocab = build_vocab(cfg["data"], model_dir=model_dir)52    53    model = build_model(cfg["model"], src_vocab=src_vocab, trg_vocab=trg_vocab)54    55    # load model state from disk56    model_checkpoint = load_checkpoint(ckpt, device=device)57    model.load_state_dict(model_checkpoint["model_state"])58    59    if device.type == "cuda":60        model.to(device)61    62    tokenizer = build_tokenizer(cfg["data"])63    sequence_encoder = {64        src_cfg["lang"]: partial(src_vocab.sentences_to_ids, bos=False, eos=True),65        trg_cfg["lang"]: None,66    }67    68    test_cfg["batch_size"] = 1  # CAUTION: this will raise an error if n_gpus > 169    test_cfg["batch_type"] = "sentence"70    71    test_data = build_dataset(72        dataset_type="stream",73        path=None,74        src_lang=src_cfg["lang"],75        trg_lang=trg_cfg["lang"],76        split="test",77        tokenizer=tokenizer,78        sequence_encoder=sequence_encoder,79    )80    test_data.set_item(text.strip())81 82    cfg=test_cfg83    _, _, hypotheses, trg_tokens, trg_scores, _ = predict(84        model=model,85        data=test_data,86        compute_loss=False,87        device=device,88        n_gpu=n_gpu,89        normalization="none",90        num_workers=num_workers,91        cfg=cfg,92        fp16=fp16,93    )94    return hypotheses[0]95 96title = """97<center><strong><font size='8'>Script Normalization for Unconventional Writing<font></strong></center>98 99<div align="center">100    <img src="https://raw.githubusercontent.com/sinaahmadi/ScriptNormalization/main/Perso-Arabic_scripts.jpg" alt="Perso-Arabic scripts used by the target languages in our paper" width="400">101</div>102 103<h3 style="font-weight: 450; font-size: 1rem; margin: 0rem"> 104    [<a href="https://sinaahmadi.github.io/docs/articles/ahmadi2023acl.pdf" style="color:blue;">Paper (ACL 2023)</a>] 105    [<a href="https://sinaahmadi.github.io/docs/slides/ahmadi2023acl_slides.pdf" style="color:blue;">Slides</a>]106    [<a href="https://github.com/sinaahmadi/ScriptNormalization" style="color:blue;">GitHub</a>]107    [<a href="https://s3.amazonaws.com/pf-user-files-01/u-59356/uploads/2023-06-04/rw32pwp/ACL2023.mp4" style="color:blue;">Presentation</a>]108</h3>109    """110 111description = """112<ul>113    <li style="font-size:120%;">&quot;<em>mar7aba!</em>&quot;</li>114    <li style="font-size:120%;">&quot;<em>هاو ئار یوو؟</em>&quot;</li>115    <li style="font-size:120%;">&quot;<em>Μπιάνβενου α σετ ντεμό!</em>&quot;</li>116</ul>117 118<p style="font-size:120%;">What do all these sentences have in common?  Being greeted in Arabic with &quot;<em>mar7aba</em>&quot; written in the Latin script, then asked how you are (&quot;<em>هاو ئار یوو؟</em>&quot;) in English using the Perso-Arabic script of Kurdish and then, welcomed to this demo in French (&quot;<em>Μπιάνβενου α σετ ντεμό!</em>&quot;) written in Greek script. All these sentences are written in an <strong>unconventional</strong> script.</p>119 120<p style="font-size:120%;">Although you may find these sentences risible, unconventional writing is a common practice among millions of speakers in bilingual communities. In our paper entitled &quot;<a href="https://sinaahmadi.github.io/docs/articles/ahmadi2023acl.pdf" target="_blank"><strong>Script Normalization for Unconventional Writing of Under-Resourced Languages in Bilingual Communities</strong></a>&quot;, we shed light on this problem and propose an approach to normalize noisy text written in unconventional writing.</p>121 122<p style="font-size:120%;">This demo deploys a few models that are trained for <strong>the normalization of unconventional writing</strong>. Please note that this tool is not a spell-checker and cannot correct errors beyond character normalization. For better performance, you can apply hard-coded rules on the input and then pass it to the models, hence a hybrid system.</p>123 124<p style="font-size:120%;">For more information, you can check out the project on GitHub too: <a href="https://github.com/sinaahmadi/ScriptNormalization" target="_blank"><strong>https://github.com/sinaahmadi/ScriptNormalization</strong></a></p>125"""126 127examples = [128    ["بو شهرین نوفوسو ، 2014 نجی ایلين نوفوس ساییمی اساسيندا 41 نفر ایمیش .", "Azeri Turkish in Persian"],#"بۇ شهرین نۆفوسو ، 2014 نجی ایلين نۆفوس ساییمی اساسيندا 41 نفر ایمیش ."129    ["ياخوا تةمةن دريژبيت بوئةم ميللةتة", "Central Kurdish in Arabic"],130    ["یکیک له جوانیکانی ام شاره جوانه", "Central Kurdish in Persian"],131    ["نمک درهٰ مردوم گيلک ايسن ؤ اوشان زوان ني گيلکي ايسه .", "Gilaki in Persian"],132    ["شؤنةو اانةيةرة گةشت و گلي ناجارانةو اؤجالاني دةستش پنةكةرد", "Gorani in Arabic"], #شۆنەو ئانەیەرە گەشت و گێڵی ناچارانەو ئۆجالانی دەستش پنەکەرد133    ["ڕوٙو زوانی ئەذایی چەنی پەیذابی ؟", "Gorani in Central Kurdish"], # ڕوٙو زوانی ئەڎایی چەنی پەیڎابی ؟134    ["هنگامکان ظميٛ ر چمان ، بپا کريٛلي بيشان :", "Gorani in Persian"], # هەنگامەکان وزمیٛ وەرو چەمان ، بەپاو کریٛڵی بیەشان :135    ["ربعی بن افکل اُسے اَکھ صُحابی .", "Kashmiri in Urdu"], # ربعی بن افکل ٲسؠ اَکھ صُحابی .136    ["اینتا زون گنشکرون 85 میلیون نفر هسن", "Mazandarani in Persian"], # اینتا زوون گِنِشکَرون 85 میلیون نفر هسنه137    ["بة رطكا هة صطئن ژ دل هاطة  بة لافكرن", "Northern Kurdish in Arabic"], #پەرتوکا هەستێن ژ دل هاتە بەلافکرن138    ["ثرکى همرنگ نرميني دويت هندک قوناغين دي ببريت", "Northern Kurdish in Persian"], # سەرەکی هەمەرەنگ نەرمینێ دڤێت هندەک قوناغێن دی ببڕیت139    ["ہتی کجھ اپ ۽ تمام دائون ترینون بیھندیون آھن .", "Sindhi in Urdu"] # هتي ڪجھ اپ ۽ تمام ڊائون ٽرينون بيھنديون آھن .140]141 142 143article =  """144<div style="text-align: justify; max-width: 1200px; margin: 20px auto;">145    <h3 style="font-weight: 450; font-size: 1rem; margin: 0rem">146        <b>Created and deployed by Sina Ahmadi <a href="https://sinaahmadi.github.io/">(https://sinaahmadi.github.io/)</a>.147    </h3>148</div>149    """150demo = gr.Interface(151    title=title,152    description=description,153    fn=normalize,154    inputs=[155        gr.Textbox(lines=4, label="Noisy Text \U0001F974"),156        gr.Dropdown(label="Language in unconventional script", choices=sorted(list(languages_scripts.keys()))),157    ],158    outputs=gr.Textbox(label="Normalized Text \U0001F642"),159    examples=examples,160    article=article,161    examples_per_page=20,162    cache_examples=False  163)164 165demo.launch()166