Team Ai
Apppublic

sumanthd/IndicTrans-MultilingualTranslation

sourceHugging Facemitupdated 4y agoView on Hugging Face
6likes
postprocess_translate.py111 linesDownload Raw Back to scripts
1INDIC_NLP_LIB_HOME = "indic_nlp_library"2INDIC_NLP_RESOURCES = "indic_nlp_resources"3import sys4 5from indicnlp import transliterate6 7sys.path.append(r"{}".format(INDIC_NLP_LIB_HOME))8from indicnlp import common9 10common.set_resources_path(INDIC_NLP_RESOURCES)11from indicnlp import loader12 13loader.load()14from sacremoses import MosesPunctNormalizer15from sacremoses import MosesTokenizer16from sacremoses import MosesDetokenizer17from collections import defaultdict18 19import indicnlp20from indicnlp.tokenize import indic_tokenize21from indicnlp.tokenize import indic_detokenize22from indicnlp.normalize import indic_normalize23from indicnlp.transliterate import unicode_transliterate24 25 26def postprocess(27    infname, outfname, input_size, lang, common_lang="hi", transliterate=False28):29    """30    parse fairseq interactive output, convert script back to native Indic script (in case of Indic languages) and detokenize.31 32    infname: fairseq log file33    outfname: output file of translation (sentences not translated contain the dummy string 'DUMMY_OUTPUT'34    input_size: expected number of output sentences35    lang: language36    """37 38    consolidated_testoutput = []39    # with open(infname,'r',encoding='utf-8') as infile:40    # consolidated_testoutput= list(map(lambda x: x.strip(), filter(lambda x: x.startswith('H-'),infile) ))41    # consolidated_testoutput.sort(key=lambda x: int(x.split('\t')[0].split('-')[1]))42    # consolidated_testoutput=[ x.split('\t')[2] for x in consolidated_testoutput ]43 44    consolidated_testoutput = [(x, 0.0, "") for x in range(input_size)]45    temp_testoutput = []46    with open(infname, "r", encoding="utf-8") as infile:47        temp_testoutput = list(48            map(49                lambda x: x.strip().split("\t"),50                filter(lambda x: x.startswith("H-"), infile),51            )52        )53        temp_testoutput = list(54            map(lambda x: (int(x[0].split("-")[1]), float(x[1]), x[2]), temp_testoutput)55        )56        for sid, score, hyp in temp_testoutput:57            consolidated_testoutput[sid] = (sid, score, hyp)58        consolidated_testoutput = [x[2] for x in consolidated_testoutput]59 60    if lang == "en":61        en_detok = MosesDetokenizer(lang="en")62        with open(outfname, "w", encoding="utf-8") as outfile:63            for sent in consolidated_testoutput:64                outfile.write(en_detok.detokenize(sent.split(" ")) + "\n")65    else:66        xliterator = unicode_transliterate.UnicodeIndicTransliterator()67        with open(outfname, "w", encoding="utf-8") as outfile:68            for sent in consolidated_testoutput:69                if transliterate:70                    outstr = indic_detokenize.trivial_detokenize(71                        xliterator.transliterate(sent, common_lang, lang), lang72                    )73                else:74                    outstr = indic_detokenize.trivial_detokenize(sent, lang)75                outfile.write(outstr + "\n")76 77 78if __name__ == "__main__":79    #     # The path to the local git repo for Indic NLP library80    # INDIC_NLP_LIB_HOME="indic_nlp_library"81    # INDIC_NLP_RESOURCES = "indic_nlp_resources"82    # sys.path.append('{}'.format(INDIC_NLP_LIB_HOME))83    # common.set_resources_path(INDIC_NLP_RESOURCES)84    #     # The path to the local git repo for Indic NLP Resources85    #     INDIC_NLP_RESOURCES=""86 87    #     sys.path.append('{}'.format(INDIC_NLP_LIB_HOME))88    #     common.set_resources_path(INDIC_NLP_RESOURCES)89 90    # loader.load()91 92    infname = sys.argv[1]93    outfname = sys.argv[2]94    input_size = int(sys.argv[3])95    lang = sys.argv[4]96    if len(sys.argv) == 5:97        transliterate = False98    elif len(sys.argv) == 6:99        transliterate = sys.argv[5]100        if transliterate.lower() == "true":101            transliterate = True102        else:103            transliterate = False104    else:105        print(f"Invalid arguments: {sys.argv}")106        exit()107 108    postprocess(109        infname, outfname, input_size, lang, common_lang="hi", transliterate=transliterate110    )111