Raviteja25012003/StackOverFlow
0
1import streamlit as st2import joblib3import pandas as pd4import re5from unidecode import unidecode6import emoji7import string8import contractions9from nltk.stem import PorterStemmer10import numpy as np11from sklearn.feature_extraction.text import ENGLISH_STOP_WORDS12 13# Custom CSS for StackOverflow-like styling14st.markdown("""15<style>16 /* Main styling */17 .stApp {18 background-color: #f8f9f9;19 }20 .stTextArea textarea {21 background-color: #ffffff;22 border: 1px solid #d6d9dc;23 border-radius: 3px;24 }25 .stButton button {26 background-color: #0095ff;27 color: white;28 border: 1px solid #07c;29 border-radius: 3px;30 padding: 0.5em 1em;31 font-weight: bold;32 }33 .stButton button:hover {34 background-color: #0077cc;35 color: white;36 }37 /* Header styling */38 .header {39 background-color: #f8f9f9;40 padding: 1rem;41 border-bottom: 1px solid #d6d9dc;42 margin-bottom: 1.5rem;43 }44 .header h1 {45 color: #242729;46 font-size: 2rem;47 }48 .header p {49 color: #6a737c;50 }51 /* Result styling */52 .result-box {53 background-color: #ffffff;54 border: 1px solid #d6d9dc;55 border-radius: 3px;56 padding: 1.5rem;57 margin-top: 1.5rem;58 box-shadow: 0 1px 3px rgba(0,0,0,0.1);59 }60 .tag {61 display: inline-block;62 background-color: #e1ecf4;63 color: #39739d;64 padding: 0.4em 0.8em;65 margin: 0.2em;66 border-radius: 3px;67 font-size: 0.9rem;68 font-weight: bold;69 }70 .tag:hover {71 background-color: #d1e5f1;72 color: #2c5777;73 }74 /* Footer styling */75 .footer {76 margin-top: 3rem;77 padding-top: 1rem;78 border-top: 1px solid #d6d9dc;79 color: #6a737c;80 font-size: 0.8rem;81 }82</style>83""", unsafe_allow_html=True)84 85# Initialize NLP components86stemmer = PorterStemmer()87stop_words = set(ENGLISH_STOP_WORDS)88 89# Chat words dictionary (extended example)90chat_words = {91 "brb": "be right back", "btw": "by the way", "lol": "laugh out loud",92 "afaik": "as far as i know", "imo": "in my opinion", "tbh": "to be honest",93 "idk": "i don't know", "asap": "as soon as possible", "np": "no problem",94 "thx": "thanks", "pls": "please", "fyi": "for your information"95}96 97def preprocess_text(text):98 """Enhanced text preprocessing"""99 if not isinstance(text, str) or not text.strip():100 return ""101 102 try:103 # Basic cleaning104 text = re.sub(r'<[^>]+>', '', text) # remove HTML105 text = re.sub(r'https?://\S+|www\.\S+', '', text) # remove URLs106 text = emoji.demojize(text, delimiters=(" ", " ")) # convert emojis107 text = unidecode(text) # remove accents108 text = contractions.fix(text) # expand contractions109 text = text.lower() # lowercase110 111 # Handle chat words and punctuation112 words = text.split()113 text = " ".join([chat_words.get(word.lower(), word) for word in words])114 text = text.translate(str.maketrans('', '', string.punctuation))115 116 # Tokenization and stemming117 tokens = re.findall(r'\b\w+\b', text)118 tokens = [word for word in tokens if word not in stop_words]119 tokens = [stemmer.stem(word) for word in tokens]120 121 return " ".join(tokens)122 except Exception as e:123 st.error(f"Text preprocessing error: {e}")124 return ""125 126# Load model and label binarizer127@st.cache_resource128def load_models():129 try:130 # For Hugging Face Spaces, use relative paths131 model = joblib.load("tag_model.joblib")132 mlb = joblib.load("tag_binarizer.joblib")133 return model, mlb134 except Exception as e:135 st.error(f"Error loading models: {e}")136 return None, None137 138model, mlb = load_models()139 140# Header with StackOverflow-like design141st.markdown("""142<div class="header">143 <div style="display: flex; align-items: center; gap: 15px;">144 <img src="https://cdn.sstatic.net/Sites/stackoverflow/Img/apple-touch-icon@2.png?v=73d79a89bded" width="50" style="border-radius: 5px;">145 <div>146 <h1>StackOverflow Tag Predictor</h1>147 <p>Enter a programming question to predict relevant tags</p>148 </div>149 </div>150</div>151""", unsafe_allow_html=True)152 153# User input section154st.markdown("""155<div style="background-color: #ffffff; padding: 1.5rem; border-radius: 3px; border: 1px solid #d6d9dc;">156 <h3 style="color: #242729; margin-top: 0;">๐ Your Question</h3>157""", unsafe_allow_html=True)158 159user_input = st.text_area(160 "",161 height=200,162 placeholder="Paste your programming question here...\nExample: 'How to sort a dictionary by value in Python?'",163 label_visibility="collapsed"164)165 166st.markdown("</div>", unsafe_allow_html=True)167 168if st.button("๐ Predict Tags", type="primary"):169 if not user_input.strip():170 st.warning("Please enter some text to predict tags.")171 elif model is None or mlb is None:172 st.error("Model failed to load. Please check the deployment logs.")173 else:174 with st.spinner("Analyzing your question..."):175 processed = preprocess_text(user_input)176 if processed:177 try:178 # Create input DataFrame with the correct column name179 input_df = pd.DataFrame({'processed_excerpt': [processed]})180 181 # Get predictions182 if hasattr(model, "predict_proba"):183 # For probability-based models184 probabilities = model.predict_proba(input_df)[0]185 top5_idx = np.argsort(probabilities)[-5:][::-1]186 top5_tags = [mlb.classes_[i] for i in top5_idx]187 top5_probs = [probabilities[i] for i in top5_idx]188 189 # Display results in a styled box190 st.markdown("""191 <div class="result-box">192 <h3 style="color: #242729; margin-top: 0;">๐ท๏ธ Predicted Tags</h3>193 <p style="color: #6a737c;">Most relevant tags for your question:</p>194 """, unsafe_allow_html=True)195 196 for tag, prob in zip(top5_tags, top5_probs):197 confidence = int(prob * 100)198 st.markdown(f"""199 <div style="margin-bottom: 0.5rem;">200 <span class="tag">{tag}</span>201 <span style="color: #6a737c; font-size: 0.8rem;">(confidence: {confidence}%)</span>202 </div>203 """, unsafe_allow_html=True)204 205 st.markdown("</div>", unsafe_allow_html=True)206 207 elif hasattr(model, "decision_function"):208 # For linear models with decision function209 scores = model.decision_function(input_df)210 scores = scores[0] if isinstance(scores, (list, np.ndarray)) else scores211 top5_idx = np.argsort(scores)[-5:][::-1]212 top5_tags = [mlb.classes_[i] for i in top5_idx]213 214 st.markdown("""215 <div class="result-box">216 <h3 style="color: #242729; margin-top: 0;">๐ท๏ธ Predicted Tags</h3>217 <p style="color: #6a737c;">Most relevant tags for your question:</p>218 """, unsafe_allow_html=True)219 220 for tag in top5_tags:221 st.markdown(f'<span class="tag">{tag}</span>', unsafe_allow_html=True)222 223 st.markdown("</div>", unsafe_allow_html=True)224 225 else:226 # Fallback to simple prediction227 pred = model.predict(input_df)228 predicted_tags = mlb.inverse_transform(pred)229 230 # Flatten and clean tags231 clean_tags = []232 for tags in predicted_tags:233 if isinstance(tags, (np.ndarray, list, tuple)):234 clean_tags.extend([str(tag) for tag in tags])235 else:236 clean_tags.append(str(tags))237 238 # Remove duplicates and get top 5239 unique_tags = list(dict.fromkeys(clean_tags))240 top_tags = unique_tags[:5]241 242 if top_tags:243 st.markdown("""244 <div class="result-box">245 <h3 style="color: #242729; margin-top: 0;">๐ท๏ธ Predicted Tags</h3>246 <p style="color: #6a737c;">Most relevant tags for your question:</p>247 """, unsafe_allow_html=True)248 249 for tag in top_tags:250 st.markdown(f'<span class="tag">{tag}</span>', unsafe_allow_html=True)251 252 st.markdown("</div>", unsafe_allow_html=True)253 else:254 st.info("No tags predicted. Try with a more descriptive excerpt.")255 except Exception as e:256 st.error(f"Prediction error: {str(e)}")257 258# Footer259st.markdown("""260<div class="footer">261 <p>This tool uses machine learning to predict StackOverflow tags based on question content.</p>262 <p>Note: Predictions are based on historical StackOverflow data and may not be perfect.</p>263</div>264""", unsafe_allow_html=True)