Team Ai
Apppublic

Daextream/mmstts

sourceHugging Faceupdated 3y agoView on Hugging Face
0likes
app.py215 linesDownload Raw Back to root
1# Based on example code of https://huggingface.co/facebook/m2m100_1.2B2# and https://github.com/wannaphong/ttsmms3# See also https://github.com/facebookresearch/fairseq/blob/main/examples/mms/README.md4 5import gradio as gr6import os7import re8import soundfile as sf9 10import json11import nltk12from underthesea import sent_tokenize as vie_sent_tokenize  # Vietnamese NLP toolkit13from underthesea import text_normalize as vie_text_normalize14from nltk import sent_tokenize as nltk_sent_tokenize15from ttsmms import download16from ttsmms import TTS17 18from collections import OrderedDict19import uuid20import datetime21import shutil22from num2words import num2words23 24 25this_description = """Text To Speech for [1000+ languages](https://dl.fbaipublicfiles.com/mms/tts/all-tts-languages.html) - using [fairseq MMS TTS](https://github.com/facebookresearch/fairseq/blob/main/examples/mms/README.md) and [ttsmms](https://github.com/wannaphong/ttsmms) wrapper.26Please note that for some languages, it may not pronounce all words correctly (yet).27"""28 29nltk.download("punkt")30 31# Pre-download some languages32tts_models = {}33eng_path = download("eng", "./data")34tts_models["eng"] = eng_path35vie_path = download("vie", "./data")36tts_models["vie"] = vie_path37mya_path = download("mya", "./data")38tts_models["mya"] = mya_path39 40lang_codes = OrderedDict()41 42language_names = list(lang_codes.keys())43with open("lang_code.txt", "r") as file:44    for line in file:45        line = line.strip()46        if line.startswith("----"):47            continue48        iso, lang = line.split("\t", 1)49        lang_codes[lang + " (" + iso + ")"] = iso50 51language_names = list(lang_codes.keys())52 53# Load num2words_lang_map54with open("num2words_lang_map.json") as f:55    num2words_lang_map = json.load(f, object_pairs_hook=OrderedDict)56 57 58def convert_numbers_to_words_num2words(text, lang):59    # Find all numbers in the text using regex60    numbers = re.findall(r"\d+", text)61    # Sort numbers in descending order of length62    sorted_numbers = sorted(numbers, key=len, reverse=True)63    print(sorted_numbers)64 65    # Replace numbers with their word equivalents66    for number in sorted_numbers:67        number_word = num2words(int(number), lang=num2words_lang_map[lang][0])68        text = text.replace(number, number_word)69 70    return text71 72 73def convert_mya_numbers_to_words(text):74    from mm_num2word import mm_num2word, extract_num75 76    numbers = extract_num(text)77    sorted_numbers = sorted(numbers, key=len, reverse=True)78    print(sorted_numbers)79 80    for n in sorted_numbers:81        text = text.replace(n, mm_num2word(n))82    return text83 84 85def prepare_sentences(text, lang="mya"):86    sentences = []87    # pre-process the text for some languages88    if lang.lower() == "mya":89        text = convert_mya_numbers_to_words(text)90        text = text.replace("\u104A", ",").replace("\u104B", ".")91 92    if lang in num2words_lang_map:93        print("num2words supports this lang", lang)94        text = convert_numbers_to_words_num2words(text, lang)95    print("Processed text", text)96 97    # Not sure why this can fix unclear pronunciation for the first word of vie98    text = text.lower()99 100    paragraphs = [paragraph for paragraph in text.split("\n") if paragraph.strip()]101 102    if lang.lower() == "vie":103        for paragraph in paragraphs:104            sentences_raw = vie_sent_tokenize(paragraph)105            sentences.extend(106                [107                    vie_text_normalize(sentence)108                    for sentence in sentences_raw109                    if sentence.strip()110                ]111            )112    else:113        sentences = [114            sentence115            for paragraph in paragraphs116            for sentence in nltk_sent_tokenize(paragraph)117            if sentence.strip()118        ]119    return sentences120 121 122def list_dir(lang):123    # Get the current directory124    current_dir = os.getcwd()125    print(current_dir)126 127    # List all files in the current directory128    files = os.listdir(current_dir)129 130    # Filter the list to include only WAV files131    wav_files = [file for file in files if file.endswith(".wav")]132    print("Total wav files:", len(wav_files))133 134    # Print the last WAV file135    sorted_list = sorted(wav_files)136    print(lang, sorted_list[-1])137 138 139def combine_wav(source_dir, stamp, lang):140    # Get a list of all WAV files in the folder141    wav_files = [file for file in os.listdir(source_dir) if file.endswith(".wav")]142 143    # Sort the files alphabetically to ensure the correct order of combination144    wav_files.sort()145 146    # Combine the WAV files147    combined_data = []148    for file in wav_files:149        file_path = os.path.join(source_dir, file)150        data, sr = sf.read(file_path)151        combined_data.extend(data)152 153    # Save the combined audio to a new WAV file154    combined_file_path = f"{stamp}_{lang}.wav"155    sf.write(combined_file_path, combined_data, sr)156 157    shutil.rmtree(source_dir)158    list_dir(lang)159 160    # Display the combined audio in the Hugging Face Space app161    return combined_file_path162 163 164def mms_tts(Input_Text, lang_name="Burmese (mya)"):165    # lang_code = lang_codes[lang_name]166    try:167        lang_code = lang_codes[lang_name]168    except KeyError:169        lang_code = "mya"170 171    user_model = download(lang_code, "./data")172    tts = TTS(user_model)173 174    sentences = prepare_sentences(Input_Text, lang_code)175 176    # output_dir = f"out_{lang_code}"177    current_datetime = datetime.datetime.now()178    timestamp = current_datetime.strftime("%Y%m%d%H%M%S%f")179 180    user_dir = f"u_{timestamp}"181    if os.path.exists(user_dir):182        session_id = str(uuid.uuid4())  # Generate a random session ID183        user_dir = f"u_{session_id}_{timestamp}"184    os.makedirs(user_dir, exist_ok=True)185    print("New user directory", user_dir)186 187    for i, sentence in enumerate(sentences):188        tts.synthesis(sentence, wav_path=f"{user_dir}/s_{str(i).zfill(10)}.wav")189    combined_file_path = combine_wav(user_dir, timestamp, lang_code)190    return combined_file_path191 192 193# common_languages = ["eng", "mya", "vie"]  # List of common language codes194iface = gr.Interface(195    fn=mms_tts,196    title="Massively Multilingual Speech (MMS) - Text To Speech",197    description=this_description,198    inputs=[199        gr.Textbox(lines=5, placeholder="Enter text (unlimited sentences)", label="Input text (unlimited sentences)"),200        gr.Dropdown(201            choices=language_names,202            label="Select language 1,000+",203            value="Burmese (mya)",204        ),205    ],206    outputs="audio",207)208# outputs=[209#         "audio",210#         gr.File(label="Download", type="file", download_to="done.wav")211#     ])212 213 214iface.launch()215