mahmoudmohammad/Topic-Classification
0
1import gradio as gr2import torch3import collections4import re5from transformers import AutoTokenizer, AutoModelForSequenceClassification6 7# Camel-Tools Preprocessing Libraries8from camel_tools.utils.normalize import normalize_alef_maksura_ar9from camel_tools.utils.normalize import normalize_alef_ar10from camel_tools.utils.normalize import normalize_teh_marbuta_ar11from camel_tools.utils.dediac import dediac_ar12 13HF_USERNAME = "mahmoudmohammad" 14CONFIDENCE_THRESHOLD = 0.7015 16# --- 0. Exact Same Preprocessing used in Training Phase ---17def clean_arabic_news(text):18 if not isinstance(text, str): return ""19 # Strip garbage characters20 text = re.sub(r'http\S+|www.\S+', '', text)21 text = re.sub(r'<.*?>', '', text) 22 text = re.sub(r'@\w+', '', text) 23 text = re.sub(r'\s+', ' ', text).strip()24 25 # NLP Morphology standardization26 text = dediac_ar(text)27 text = normalize_alef_ar(text)28 text = normalize_alef_maksura_ar(text)29 text = normalize_teh_marbuta_ar(text)30 return text31 32print("Booting Global Taxonomy Engine...")33 34# --- 1. Permanently Load L1 Model ---35l1_repo = f"{HF_USERNAME}/SANAD-L1-Root-Classifier"36l1_tokenizer = AutoTokenizer.from_pretrained(l1_repo)37l1_model = AutoModelForSequenceClassification.from_pretrained(l1_repo)38l1_model.eval()39 40# --- 2. Smart Memory Manager (LRU Cache) ---41class L2ModelCache:42 def __init__(self, max_models=3):43 self.max_models = max_models44 self.cache = collections.OrderedDict()45 46 def get_model(self, l1_label):47 if l1_label in self.cache:48 self.cache.move_to_end(l1_label)49 return self.cache[l1_label]50 51 print(f"Loading {l1_label} L2 model into RAM...")52 repo_id = f"{HF_USERNAME}/SANAD-L2-{l1_label}-Classifier"53 54 try:55 tok = AutoTokenizer.from_pretrained(repo_id)56 mod = AutoModelForSequenceClassification.from_pretrained(repo_id)57 mod.eval()58 self.cache[l1_label] = (tok, mod)59 60 if len(self.cache) > self.max_models:61 evicted = self.cache.popitem(last=False)62 print(f"Unloaded {evicted[0]} L2 model from RAM.")63 return self.cache[l1_label]64 except Exception:65 return None, None 66 67l2_manager = L2ModelCache(max_models=3)68 69# --- 3. The 2-Stage Routing Logic ---70def classify_news(text):71 if not text.strip():72 return "Empty text", "N/A"73 74 # CRITICAL: Clean the incoming API request!75 cleaned_text = clean_arabic_news(text)76 77 # Stage 1: L1 Routing78 inputs = l1_tokenizer(cleaned_text, return_tensors="pt", truncation=True, max_length=256)79 with torch.no_grad():80 out1 = l1_model(**inputs)81 82 probs1 = torch.softmax(out1.logits, dim=-1).squeeze()83 conf1 = probs1.max().item()84 pred1 = l1_model.config.id2label[probs1.argmax().item()]85 86 if conf1 < CONFIDENCE_THRESHOLD:87 return "Uncertain", f"L1 Drop: {pred1} (Conf: {conf1:.2f})"88 89 l2_tok, l2_mod = l2_manager.get_model(pred1)90 91 if not l2_mod:92 return pred1, f"Status: L1 Flat Structure Approved (Conf: {conf1:.2f})"93 94 # Stage 2: Ensure we feed the CLEAN text here as well95 l2_in = l2_tok(cleaned_text, return_tensors="pt", truncation=True, max_length=256)96 with torch.no_grad():97 out2 = l2_mod(**l2_in)98 99 probs2 = torch.softmax(out2.logits, dim=-1).squeeze()100 conf2 = probs2.max().item()101 pred2 = l2_mod.config.id2label[probs2.argmax().item()]102 103 if conf2 < CONFIDENCE_THRESHOLD:104 return pred1, f"Status: Sub-Tag Rejected. Dropped to Root (L2 Conf: {conf2:.2f})"105 106 return f"{pred1} / {pred2}", f"Success: L1({conf1:.2f}) -> L2({conf2:.2f})"107 108# --- 4. The Front-End UI ---109iface = gr.Interface(110 fn=classify_news,111 inputs=gr.Textbox(lines=7, label="Arabic News Text", placeholder="Paste article here..."),112 outputs=[113 gr.Textbox(label="Final Category Assignment"), 114 gr.Textbox(label="Confidence Diagnostics")115 ],116 title="Arabic News Hierarchical Categorizer (L1 + L2 Pipeline)",117 description="This gateway intelligently filters, normalizes, and classifies Arabic text dynamically.",118 examples=["سجل فريق ريال مدريد فوزاً كاسحاً في دوري أبطال أوروبا"]119)120 121iface.launch()