Team Ai
Apppublic

Raviteja25012003/StackOverFlow

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
app.py264 linesDownload Raw Back to root
1import streamlit as st2import joblib3import pandas as pd4import re5from unidecode import unidecode6import emoji7import string8import contractions9from nltk.stem import PorterStemmer10import numpy as np11from sklearn.feature_extraction.text import ENGLISH_STOP_WORDS12 13# Custom CSS for StackOverflow-like styling14st.markdown("""15<style>16    /* Main styling */17    .stApp {18        background-color: #f8f9f9;19    }20    .stTextArea textarea {21        background-color: #ffffff;22        border: 1px solid #d6d9dc;23        border-radius: 3px;24    }25    .stButton button {26        background-color: #0095ff;27        color: white;28        border: 1px solid #07c;29        border-radius: 3px;30        padding: 0.5em 1em;31        font-weight: bold;32    }33    .stButton button:hover {34        background-color: #0077cc;35        color: white;36    }37    /* Header styling */38    .header {39        background-color: #f8f9f9;40        padding: 1rem;41        border-bottom: 1px solid #d6d9dc;42        margin-bottom: 1.5rem;43    }44    .header h1 {45        color: #242729;46        font-size: 2rem;47    }48    .header p {49        color: #6a737c;50    }51    /* Result styling */52    .result-box {53        background-color: #ffffff;54        border: 1px solid #d6d9dc;55        border-radius: 3px;56        padding: 1.5rem;57        margin-top: 1.5rem;58        box-shadow: 0 1px 3px rgba(0,0,0,0.1);59    }60    .tag {61        display: inline-block;62        background-color: #e1ecf4;63        color: #39739d;64        padding: 0.4em 0.8em;65        margin: 0.2em;66        border-radius: 3px;67        font-size: 0.9rem;68        font-weight: bold;69    }70    .tag:hover {71        background-color: #d1e5f1;72        color: #2c5777;73    }74    /* Footer styling */75    .footer {76        margin-top: 3rem;77        padding-top: 1rem;78        border-top: 1px solid #d6d9dc;79        color: #6a737c;80        font-size: 0.8rem;81    }82</style>83""", unsafe_allow_html=True)84 85# Initialize NLP components86stemmer = PorterStemmer()87stop_words = set(ENGLISH_STOP_WORDS)88 89# Chat words dictionary (extended example)90chat_words = {91    "brb": "be right back", "btw": "by the way", "lol": "laugh out loud",92    "afaik": "as far as i know", "imo": "in my opinion", "tbh": "to be honest",93    "idk": "i don't know", "asap": "as soon as possible", "np": "no problem",94    "thx": "thanks", "pls": "please", "fyi": "for your information"95}96 97def preprocess_text(text):98    """Enhanced text preprocessing"""99    if not isinstance(text, str) or not text.strip():100        return ""101    102    try:103        # Basic cleaning104        text = re.sub(r'<[^>]+>', '', text)  # remove HTML105        text = re.sub(r'https?://\S+|www\.\S+', '', text)  # remove URLs106        text = emoji.demojize(text, delimiters=(" ", " "))  # convert emojis107        text = unidecode(text)  # remove accents108        text = contractions.fix(text)  # expand contractions109        text = text.lower()  # lowercase110        111        # Handle chat words and punctuation112        words = text.split()113        text = " ".join([chat_words.get(word.lower(), word) for word in words])114        text = text.translate(str.maketrans('', '', string.punctuation))115        116        # Tokenization and stemming117        tokens = re.findall(r'\b\w+\b', text)118        tokens = [word for word in tokens if word not in stop_words]119        tokens = [stemmer.stem(word) for word in tokens]120        121        return " ".join(tokens)122    except Exception as e:123        st.error(f"Text preprocessing error: {e}")124        return ""125 126# Load model and label binarizer127@st.cache_resource128def load_models():129    try:130        # For Hugging Face Spaces, use relative paths131        model = joblib.load("tag_model.joblib")132        mlb = joblib.load("tag_binarizer.joblib")133        return model, mlb134    except Exception as e:135        st.error(f"Error loading models: {e}")136        return None, None137 138model, mlb = load_models()139 140# Header with StackOverflow-like design141st.markdown("""142<div class="header">143    <div style="display: flex; align-items: center; gap: 15px;">144        <img src="https://cdn.sstatic.net/Sites/stackoverflow/Img/apple-touch-icon@2.png?v=73d79a89bded" width="50" style="border-radius: 5px;">145        <div>146            <h1>StackOverflow Tag Predictor</h1>147            <p>Enter a programming question to predict relevant tags</p>148        </div>149    </div>150</div>151""", unsafe_allow_html=True)152 153# User input section154st.markdown("""155<div style="background-color: #ffffff; padding: 1.5rem; border-radius: 3px; border: 1px solid #d6d9dc;">156    <h3 style="color: #242729; margin-top: 0;">๐Ÿ“ Your Question</h3>157""", unsafe_allow_html=True)158 159user_input = st.text_area(160    "",161    height=200,162    placeholder="Paste your programming question here...\nExample: 'How to sort a dictionary by value in Python?'",163    label_visibility="collapsed"164)165 166st.markdown("</div>", unsafe_allow_html=True)167 168if st.button("๐Ÿ” Predict Tags", type="primary"):169    if not user_input.strip():170        st.warning("Please enter some text to predict tags.")171    elif model is None or mlb is None:172        st.error("Model failed to load. Please check the deployment logs.")173    else:174        with st.spinner("Analyzing your question..."):175            processed = preprocess_text(user_input)176            if processed:177                try:178                    # Create input DataFrame with the correct column name179                    input_df = pd.DataFrame({'processed_excerpt': [processed]})180                    181                    # Get predictions182                    if hasattr(model, "predict_proba"):183                        # For probability-based models184                        probabilities = model.predict_proba(input_df)[0]185                        top5_idx = np.argsort(probabilities)[-5:][::-1]186                        top5_tags = [mlb.classes_[i] for i in top5_idx]187                        top5_probs = [probabilities[i] for i in top5_idx]188                        189                        # Display results in a styled box190                        st.markdown("""191                        <div class="result-box">192                            <h3 style="color: #242729; margin-top: 0;">๐Ÿท๏ธ Predicted Tags</h3>193                            <p style="color: #6a737c;">Most relevant tags for your question:</p>194                        """, unsafe_allow_html=True)195                        196                        for tag, prob in zip(top5_tags, top5_probs):197                            confidence = int(prob * 100)198                            st.markdown(f"""199                            <div style="margin-bottom: 0.5rem;">200                                <span class="tag">{tag}</span>201                                <span style="color: #6a737c; font-size: 0.8rem;">(confidence: {confidence}%)</span>202                            </div>203                            """, unsafe_allow_html=True)204                        205                        st.markdown("</div>", unsafe_allow_html=True)206                        207                    elif hasattr(model, "decision_function"):208                        # For linear models with decision function209                        scores = model.decision_function(input_df)210                        scores = scores[0] if isinstance(scores, (list, np.ndarray)) else scores211                        top5_idx = np.argsort(scores)[-5:][::-1]212                        top5_tags = [mlb.classes_[i] for i in top5_idx]213                        214                        st.markdown("""215                        <div class="result-box">216                            <h3 style="color: #242729; margin-top: 0;">๐Ÿท๏ธ Predicted Tags</h3>217                            <p style="color: #6a737c;">Most relevant tags for your question:</p>218                        """, unsafe_allow_html=True)219                        220                        for tag in top5_tags:221                            st.markdown(f'<span class="tag">{tag}</span>', unsafe_allow_html=True)222                        223                        st.markdown("</div>", unsafe_allow_html=True)224                        225                    else:226                        # Fallback to simple prediction227                        pred = model.predict(input_df)228                        predicted_tags = mlb.inverse_transform(pred)229                        230                        # Flatten and clean tags231                        clean_tags = []232                        for tags in predicted_tags:233                            if isinstance(tags, (np.ndarray, list, tuple)):234                                clean_tags.extend([str(tag) for tag in tags])235                            else:236                                clean_tags.append(str(tags))237                        238                        # Remove duplicates and get top 5239                        unique_tags = list(dict.fromkeys(clean_tags))240                        top_tags = unique_tags[:5]241                        242                        if top_tags:243                            st.markdown("""244                            <div class="result-box">245                                <h3 style="color: #242729; margin-top: 0;">๐Ÿท๏ธ Predicted Tags</h3>246                                <p style="color: #6a737c;">Most relevant tags for your question:</p>247                            """, unsafe_allow_html=True)248                            249                            for tag in top_tags:250                                st.markdown(f'<span class="tag">{tag}</span>', unsafe_allow_html=True)251                            252                            st.markdown("</div>", unsafe_allow_html=True)253                        else:254                            st.info("No tags predicted. Try with a more descriptive excerpt.")255                except Exception as e:256                    st.error(f"Prediction error: {str(e)}")257 258# Footer259st.markdown("""260<div class="footer">261    <p>This tool uses machine learning to predict StackOverflow tags based on question content.</p>262    <p>Note: Predictions are based on historical StackOverflow data and may not be perfect.</p>263</div>264""", unsafe_allow_html=True)