anitalp/NLP_Models_sequence
1
1import gradio as gr2from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, pipeline3 4# 1. Load Translation Model & Tokenizer Manually5model_name = "Helsinki-NLP/opus-mt-es-en"6tokenizer = AutoTokenizer.from_pretrained(model_name)7model = AutoModelForSeq2SeqLM.from_pretrained(model_name)8 9# 2. Create the translation pipeline with the explicit model/tokenizer10translator_pipe = pipeline("translation", model=model, tokenizer=tokenizer)11 12# 3. Toxicity pipeline (this one usually has no issues with the generic task)13toxicity_pipe = pipeline("text-classification", model="SkolkovoInstitute/roberta_toxicity_classifier")14 15def spanish_toxicity_check(text):16 # Step 1: Translate17 # We specify max_length to ensure it doesn't cut off long lyrics18 translation = translator_pipe(text, max_length=512)[0]['translation_text']19 20 # Step 2: Classify21 results = toxicity_pipe(translation)22 23 # Step 3: Format output for gr.Label24 return {item['label']: item['score'] for item in results}25 26# 4. Interface27demo = gr.Interface(28 fn=spanish_toxicity_check,29 inputs=gr.Textbox(label="Lyrics en Español", placeholder="Escribe aquí..."),30 outputs=gr.Label(label="Nivel de Toxicidad"),31 title="Análisis de Toxicidad de Canciones",32 description="Traducción automática (Helsinki-NLP) + Clasificación (RoBERTa)"33)34 35if __name__ == "__main__":36 demo.launch()