Team Ai
Apppublic

anitalp/NLP_Models_sequence

sourceHugging Faceupdated 8mo agoView on Hugging Face
1likes
app.py36 linesDownload Raw Back to root
1import gradio as gr2from transformers import AutoTokenizer, AutoModelForSeq2SeqLM, pipeline3 4# 1. Load Translation Model & Tokenizer Manually5model_name = "Helsinki-NLP/opus-mt-es-en"6tokenizer = AutoTokenizer.from_pretrained(model_name)7model = AutoModelForSeq2SeqLM.from_pretrained(model_name)8 9# 2. Create the translation pipeline with the explicit model/tokenizer10translator_pipe = pipeline("translation", model=model, tokenizer=tokenizer)11 12# 3. Toxicity pipeline (this one usually has no issues with the generic task)13toxicity_pipe = pipeline("text-classification", model="SkolkovoInstitute/roberta_toxicity_classifier")14 15def spanish_toxicity_check(text):16    # Step 1: Translate17    # We specify max_length to ensure it doesn't cut off long lyrics18    translation = translator_pipe(text, max_length=512)[0]['translation_text']19    20    # Step 2: Classify21    results = toxicity_pipe(translation)22    23    # Step 3: Format output for gr.Label24    return {item['label']: item['score'] for item in results}25 26# 4. Interface27demo = gr.Interface(28    fn=spanish_toxicity_check,29    inputs=gr.Textbox(label="Lyrics en Español", placeholder="Escribe aquí..."),30    outputs=gr.Label(label="Nivel de Toxicidad"),31    title="Análisis de Toxicidad de Canciones",32    description="Traducción automática (Helsinki-NLP) + Clasificación (RoBERTa)"33)34 35if __name__ == "__main__":36    demo.launch()