Daextream/mmstts
0
1# Based on example code of https://huggingface.co/facebook/m2m100_1.2B2# and https://github.com/wannaphong/ttsmms3# See also https://github.com/facebookresearch/fairseq/blob/main/examples/mms/README.md4 5import gradio as gr6import os7import re8import soundfile as sf9 10import json11import nltk12from underthesea import sent_tokenize as vie_sent_tokenize # Vietnamese NLP toolkit13from underthesea import text_normalize as vie_text_normalize14from nltk import sent_tokenize as nltk_sent_tokenize15from ttsmms import download16from ttsmms import TTS17 18from collections import OrderedDict19import uuid20import datetime21import shutil22from num2words import num2words23 24 25this_description = """Text To Speech for [1000+ languages](https://dl.fbaipublicfiles.com/mms/tts/all-tts-languages.html) - using [fairseq MMS TTS](https://github.com/facebookresearch/fairseq/blob/main/examples/mms/README.md) and [ttsmms](https://github.com/wannaphong/ttsmms) wrapper.26Please note that for some languages, it may not pronounce all words correctly (yet).27"""28 29nltk.download("punkt")30 31# Pre-download some languages32tts_models = {}33eng_path = download("eng", "./data")34tts_models["eng"] = eng_path35vie_path = download("vie", "./data")36tts_models["vie"] = vie_path37mya_path = download("mya", "./data")38tts_models["mya"] = mya_path39 40lang_codes = OrderedDict()41 42language_names = list(lang_codes.keys())43with open("lang_code.txt", "r") as file:44 for line in file:45 line = line.strip()46 if line.startswith("----"):47 continue48 iso, lang = line.split("\t", 1)49 lang_codes[lang + " (" + iso + ")"] = iso50 51language_names = list(lang_codes.keys())52 53# Load num2words_lang_map54with open("num2words_lang_map.json") as f:55 num2words_lang_map = json.load(f, object_pairs_hook=OrderedDict)56 57 58def convert_numbers_to_words_num2words(text, lang):59 # Find all numbers in the text using regex60 numbers = re.findall(r"\d+", text)61 # Sort numbers in descending order of length62 sorted_numbers = sorted(numbers, key=len, reverse=True)63 print(sorted_numbers)64 65 # Replace numbers with their word equivalents66 for number in sorted_numbers:67 number_word = num2words(int(number), lang=num2words_lang_map[lang][0])68 text = text.replace(number, number_word)69 70 return text71 72 73def convert_mya_numbers_to_words(text):74 from mm_num2word import mm_num2word, extract_num75 76 numbers = extract_num(text)77 sorted_numbers = sorted(numbers, key=len, reverse=True)78 print(sorted_numbers)79 80 for n in sorted_numbers:81 text = text.replace(n, mm_num2word(n))82 return text83 84 85def prepare_sentences(text, lang="mya"):86 sentences = []87 # pre-process the text for some languages88 if lang.lower() == "mya":89 text = convert_mya_numbers_to_words(text)90 text = text.replace("\u104A", ",").replace("\u104B", ".")91 92 if lang in num2words_lang_map:93 print("num2words supports this lang", lang)94 text = convert_numbers_to_words_num2words(text, lang)95 print("Processed text", text)96 97 # Not sure why this can fix unclear pronunciation for the first word of vie98 text = text.lower()99 100 paragraphs = [paragraph for paragraph in text.split("\n") if paragraph.strip()]101 102 if lang.lower() == "vie":103 for paragraph in paragraphs:104 sentences_raw = vie_sent_tokenize(paragraph)105 sentences.extend(106 [107 vie_text_normalize(sentence)108 for sentence in sentences_raw109 if sentence.strip()110 ]111 )112 else:113 sentences = [114 sentence115 for paragraph in paragraphs116 for sentence in nltk_sent_tokenize(paragraph)117 if sentence.strip()118 ]119 return sentences120 121 122def list_dir(lang):123 # Get the current directory124 current_dir = os.getcwd()125 print(current_dir)126 127 # List all files in the current directory128 files = os.listdir(current_dir)129 130 # Filter the list to include only WAV files131 wav_files = [file for file in files if file.endswith(".wav")]132 print("Total wav files:", len(wav_files))133 134 # Print the last WAV file135 sorted_list = sorted(wav_files)136 print(lang, sorted_list[-1])137 138 139def combine_wav(source_dir, stamp, lang):140 # Get a list of all WAV files in the folder141 wav_files = [file for file in os.listdir(source_dir) if file.endswith(".wav")]142 143 # Sort the files alphabetically to ensure the correct order of combination144 wav_files.sort()145 146 # Combine the WAV files147 combined_data = []148 for file in wav_files:149 file_path = os.path.join(source_dir, file)150 data, sr = sf.read(file_path)151 combined_data.extend(data)152 153 # Save the combined audio to a new WAV file154 combined_file_path = f"{stamp}_{lang}.wav"155 sf.write(combined_file_path, combined_data, sr)156 157 shutil.rmtree(source_dir)158 list_dir(lang)159 160 # Display the combined audio in the Hugging Face Space app161 return combined_file_path162 163 164def mms_tts(Input_Text, lang_name="Burmese (mya)"):165 # lang_code = lang_codes[lang_name]166 try:167 lang_code = lang_codes[lang_name]168 except KeyError:169 lang_code = "mya"170 171 user_model = download(lang_code, "./data")172 tts = TTS(user_model)173 174 sentences = prepare_sentences(Input_Text, lang_code)175 176 # output_dir = f"out_{lang_code}"177 current_datetime = datetime.datetime.now()178 timestamp = current_datetime.strftime("%Y%m%d%H%M%S%f")179 180 user_dir = f"u_{timestamp}"181 if os.path.exists(user_dir):182 session_id = str(uuid.uuid4()) # Generate a random session ID183 user_dir = f"u_{session_id}_{timestamp}"184 os.makedirs(user_dir, exist_ok=True)185 print("New user directory", user_dir)186 187 for i, sentence in enumerate(sentences):188 tts.synthesis(sentence, wav_path=f"{user_dir}/s_{str(i).zfill(10)}.wav")189 combined_file_path = combine_wav(user_dir, timestamp, lang_code)190 return combined_file_path191 192 193# common_languages = ["eng", "mya", "vie"] # List of common language codes194iface = gr.Interface(195 fn=mms_tts,196 title="Massively Multilingual Speech (MMS) - Text To Speech",197 description=this_description,198 inputs=[199 gr.Textbox(lines=5, placeholder="Enter text (unlimited sentences)", label="Input text (unlimited sentences)"),200 gr.Dropdown(201 choices=language_names,202 label="Select language 1,000+",203 value="Burmese (mya)",204 ),205 ],206 outputs="audio",207)208# outputs=[209# "audio",210# gr.File(label="Download", type="file", download_to="done.wav")211# ])212 213 214iface.launch()215 