roygryan/Text_Processor
0
1# import sentencepiece before transformers to avoid crushes2import sentencepiece3 4# for text generation5from transformers import pipeline 6generator = pipeline("text-generation", model="distilgpt2") # remember to add length and returning numbers7 8# for NER9ner = pipeline("ner", grouped_entities=True) # usage: ner('text')10 11# for summarization12summarizer = pipeline("summarization") # usage: summarizer('text')13 14# for POS tagging 15import nltk16nltk.download('punkt')17nltk.download('averaged_perceptron_tagger')18nltk.download('brown')19 20from textblob import TextBlob # blob = TextBlob(text) \n POS_List = blob.tags21 22# for translation 23 24from transformers import AutoTokenizer, AutoModelForSeq2SeqLM25 26tokenizer = AutoTokenizer.from_pretrained("Helsinki-NLP/opus-mt-zh-en", use_fast = False)27 28model = AutoModelForSeq2SeqLM.from_pretrained("Helsinki-NLP/opus-mt-zh-en")29 30import jionlp as jio #simtext = jio.tra2sim(tra_text, mode='char')31 32import gradio as gr33 34def TextProcessor(txt):35 # ASCII code greater than 122 will be zh36 if ord(str(txt)[0]) > 122:37 # convert to zh_sim38 sim_text = jio.tra2sim(txt, mode='char')39 zh2en_trans = pipeline("translation_zh_to_en", model = model, tokenizer = tokenizer)40 results = zh2en_trans(sim_text)[0]['translation_text']41 # ASCII code less than 122 will be en42 else:43 # if length greater than 1, sentences; otherwise, words44 if len(txt.split()) < 2:45 blob = TextBlob(txt)46 POS_List = blob.tags47 results = POS_List[0][1]48 else:49 # if txt contains ..., do text generation; otherwise do summary, NER, noun and verb phrases50 if "..." or "…" in str(txt):51 txt = str(txt)52 text = txt[0:-3]53 txt_generation = generator(text, max_length = 50, num_return_sequences = 1)54 results = txt_generation[0]["generated_text"]55 else:56 txt = str(txt)57 #txt_summarization = summarizer(txt)58 #result_01 = txt_summarization[0]59 60 #result_02 = ner(txt)61 62 blob = TextBlob(txt)63 POS_List = blob.tags64 65 noun_phrases = [np for np in POS_List if "N" in np[1][0]]66 result_03 = noun_phrases67 68 verb_phrases = [vp for vp in POS_List if "V" in vp[1][0]]69 result_04 = verb_phrases70 results = ("noun_phrases:", result_03, "verb_phrases:", result_04)71 #"Summary:", result_01['summary_text'], "NER:", result_02, 72 return results73 74final = gr.Interface(fn = TextProcessor, inputs = "text", outputs = "text")75final.launch()