Team Ai
Apppublic

Atayaz/User-Profiling-and-Segmentation

sourceHugging Faceupdated 1y agoView on Hugging Face
0likes
app.py218 linesDownload Raw Back to src
1# ------------------------------------------------------------2# 👥 User Profiling & Segmentation — Streamlit + Model Save3# ------------------------------------------------------------4# Hugging Face Spaces için:5# your-space/6# └─ src/7#    ├─ app.py8#    └─ data/user_profiles_for_ads.csv9# ------------------------------------------------------------10 11import os12import json13import numpy as np14import pandas as pd15import streamlit as st16import seaborn as sns17import matplotlib.pyplot as plt18import plotly.graph_objects as go19from collections import Counter20 21from sklearn.preprocessing import StandardScaler, OneHotEncoder22from sklearn.compose import ColumnTransformer23from sklearn.pipeline import Pipeline24from sklearn.cluster import KMeans25import joblib26 27# ================= CONFIG =================28st.set_page_config(page_title="User Profiling & Segmentation", page_icon="👥", layout="wide")29 30DATA_PATH = os.path.join("src", "data", "user_profiles_for_ads.csv")31MODELS_DIR = os.path.join("src", "models")32MODEL_NAME = "user_profiling_kmeans"33MODEL_PATH = os.path.join(MODELS_DIR, f"{MODEL_NAME}.joblib")34META_PATH = os.path.join(MODELS_DIR, f"{MODEL_NAME}.meta.json")35os.makedirs(MODELS_DIR, exist_ok=True)36 37# ================= HEADER =================38st.markdown("""39<style>40.big-hero {41  padding: 1.0rem 1.2rem;42  border-radius: 14px;43  background: linear-gradient(120deg,#06b6d4 0%,#3b82f6 50%,#8b5cf6 100%);44  color: white;45  margin-bottom: 1rem;46}47.badge { display:inline-block; padding:.2rem .6rem; border-radius:999px;48         border:1px solid rgba(255,255,255,.25); background:rgba(255,255,255,.15);49         margin-right:.4rem; font-size:.8rem; }50</style>51<div class="big-hero">52  <h2 style="margin:0;">👥 User Profiling & Segmentation</h2>53  <div style="margin-top:.4rem;">54    <span class="badge">KMeans</span>55    <span class="badge">Pipeline</span>56    <span class="badge">Radar Chart</span>57  </div>58</div>59""", unsafe_allow_html=True)60 61# ================= HELPERS =================62@st.cache_data(show_spinner=False)63def load_data(path):64    if not os.path.exists(path):65        st.error(f"Veri dosyası bulunamadı: {path}")66        st.stop()67    df = pd.read_csv(path)68    df = df.dropna()69    return df70 71def train_and_save(df, n_clusters=5):72    features = ['Age', 'Gender', 'Income Level',73                'Time Spent Online (hrs/weekday)',74                'Time Spent Online (hrs/weekend)',75                'Likes and Reactions',76                'Click-Through Rates (CTR)']77    X = df[features]78 79    numeric_features = ['Time Spent Online (hrs/weekday)',80                        'Time Spent Online (hrs/weekend)',81                        'Likes and Reactions',82                        'Click-Through Rates (CTR)']83    numeric_transformer = StandardScaler()84 85    categorical_features = ['Age', 'Gender', 'Income Level']86    categorical_transformer = OneHotEncoder()87 88    preprocessor = ColumnTransformer([89        ('num', numeric_transformer, numeric_features),90        ('cat', categorical_transformer, categorical_features)91    ])92 93    pipeline = Pipeline([94        ('preprocessor', preprocessor),95        ('cluster', KMeans(n_clusters=n_clusters, random_state=42))96    ])97 98    pipeline.fit(X)99    df['Cluster'] = pipeline.named_steps['cluster'].labels_100 101    meta = {"n_clusters": n_clusters, "inertia": float(pipeline.named_steps['cluster'].inertia_)}102    joblib.dump(pipeline, MODEL_PATH)103    with open(META_PATH, "w", encoding="utf-8") as f:104        json.dump(meta, f, indent=2)105    return df, pipeline, meta106 107def load_model():108    if not os.path.exists(MODEL_PATH):109        return None, None110    model = joblib.load(MODEL_PATH)111    meta = json.load(open(META_PATH)) if os.path.exists(META_PATH) else {}112    return model, meta113 114def bytes_of_file(path: str) -> bytes:115    with open(path, "rb") as f:116        return f.read()117 118# ================= SIDEBAR =================119with st.sidebar:120    st.header("⚙️ Ayarlar")121    n_clusters = st.slider("Küme Sayısı (KMeans)", min_value=2, max_value=10, value=5)122    retrain = st.button("🔁 Yeniden Eğit ve Kaydet")123    show_downloads = st.checkbox("Model dosyalarını göster / indir", value=True)124 125# ================= DATA LOAD =================126data = load_data(DATA_PATH)127if retrain:128    train_and_save.clear()129model, meta = load_model()130if model is None:131    st.info("Model bulunamadı, eğitim başlatılıyor...")132    data, model, meta = train_and_save(data, n_clusters)133else:134    data, _, _ = train_and_save(data, meta.get("n_clusters", n_clusters))135 136# ================= KPIs =================137c1, c2, c3 = st.columns(3)138c1.metric("Kayıt", f"{len(data):,}")139c2.metric("Küme Sayısı", f"{meta.get('n_clusters', n_clusters)}")140c3.metric("Model Inertia", f"{meta.get('inertia', 0):.2f}")141 142# ================= TABS =================143tab_eda, tab_clusters, tab_model = st.tabs(["📊 EDA", "🧠 Segment Analizi", "💾 Model Kaydet / İndir"])144 145# ----------- EDA -----------146with tab_eda:147    st.subheader("📊 Veri Keşfi ve Dağılımlar")148 149    c1, c2 = st.columns(2)150    with c1:151        st.markdown("#### Yaş ve Cinsiyet Dağılımı")152        fig, ax = plt.subplots(figsize=(6, 4))153        sns.countplot(x='Age', data=data, palette='coolwarm', ax=ax)154        ax.set_xticklabels(ax.get_xticklabels(), rotation=45)155        st.pyplot(fig)156 157    with c2:158        st.markdown("#### Eğitim ve Gelir Düzeyi Dağılımı")159        fig, ax = plt.subplots(figsize=(6, 4))160        sns.countplot(x='Income Level', data=data, palette='coolwarm', ax=ax)161        ax.set_xticklabels(ax.get_xticklabels(), rotation=45)162        st.pyplot(fig)163 164    st.markdown("#### En Popüler İlgi Alanları")165    interests_list = data['Top Interests'].str.split(', ').sum()166    interest_counts = Counter(interests_list)167    interests_df = pd.DataFrame(interest_counts.items(), columns=['Interest', 'Frequency']).sort_values(by='Frequency', ascending=False)168    fig, ax = plt.subplots(figsize=(8, 5))169    sns.barplot(y='Interest', x='Frequency', data=interests_df.head(10), palette='coolwarm', ax=ax)170    st.pyplot(fig)171 172# ----------- CLUSTERS -----------173with tab_clusters:174    st.subheader("🧠 Kullanıcı Segmentleri")175 176    numeric_features = ['Time Spent Online (hrs/weekday)', 'Time Spent Online (hrs/weekend)', 'Likes and Reactions', 'Click-Through Rates (CTR)']177    cluster_means = data.groupby('Cluster')[numeric_features].mean()178 179    # Radar chart180    features_to_plot = numeric_features181    labels = np.array(features_to_plot)182    radar_df = cluster_means.reset_index()183 184    radar_df_norm = radar_df.copy()185    for feature in features_to_plot:186        radar_df_norm[feature] = (radar_df[feature] - radar_df[feature].min()) / (radar_df[feature].max() - radar_df[feature].min())187 188    fig = go.Figure()189    for i in radar_df_norm['Cluster']:190        row = radar_df_norm[radar_df_norm['Cluster'] == i]191        fig.add_trace(go.Scatterpolar(192            r=row[features_to_plot].values.flatten().tolist() + [row[features_to_plot].values.flatten()[0]],193            theta=labels.tolist() + [labels[0]],194            fill='toself',195            name=f"Cluster {i}"196        ))197 198    fig.update_layout(199        polar=dict(radialaxis=dict(visible=True, range=[0, 1])),200        showlegend=True,201        title="User Segment Radar Chart"202    )203    st.plotly_chart(fig, use_container_width=True)204 205    st.markdown("#### Segmentlere Göre Ortalama Değerler")206    st.dataframe(cluster_means)207 208# ----------- MODEL FILES -----------209with tab_model:210    st.subheader("Model Dosyaları")211    if show_downloads and os.path.exists(MODEL_PATH):212        st.download_button("📦 Model (.joblib)", data=bytes_of_file(MODEL_PATH),213                           file_name=os.path.basename(MODEL_PATH), mime="application/octet-stream")214    if show_downloads and os.path.exists(META_PATH):215        st.download_button("📝 Meta (.json)", data=bytes_of_file(META_PATH),216                           file_name=os.path.basename(META_PATH), mime="application/json")217    st.json(meta)218