Team Ai
Apppublic

JacksonFW/transcriptomics-explorer

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
app.py738 linesDownload Raw Back to root
1"""2Transcriptomics Explorer3Interactive RNA-seq / differential expression dashboard:4  - Volcano Plot    — significant DE genes5  - MA Plot         — mean expression vs fold change6  - PCA             — sample quality check7  - Sample Corr.    — sample-to-sample Pearson correlation8  - Heatmap         — top gene expression patterns9  - Pathway Enrich. — enriched biological pathways10 11Run:  python app.py12Open: http://localhost:805013"""14 15import base6416import io17 18import dash19import dash_bootstrap_components as dbc20import numpy as np21import pandas as pd22import plotly.express as px23import plotly.graph_objects as go24from dash import Input, Output, State, dash_table, dcc, html25from scipy import stats26from scipy.cluster.hierarchy import leaves_list, linkage27 28# ─────────────────────────────────────────────29# CONSTANTS30# ─────────────────────────────────────────────31DARK   = "#0f172a"32CARD   = "#1e293b"33ACCENT = "#38bdf8"34 35PATHWAY_DB = {36    # Cell Proliferation & Cycle37    "Cell Cycle":           ["TP53","RB1","CDKN2A","CDK4","CDK6","E2F1","CCND1","CDC20","PLK1","AURKA"],38    "G1/S Transition":      ["CCND1","CDK4","CDK6","RB1","E2F1","CDKN1A","CDKN2A","CCNE1","CDK2"],39    "DNA Replication":      ["PCNA","MCM2","MCM7","RFC1","POLA1","POLE","RPA1","GINS1","CDC6"],40    # DNA Damage & Repair41    "DNA Repair":           ["BRCA1","BRCA2","ATM","MLH1","MSH2","RAD51","PARP1","ATR","CHEK1","CHEK2"],42    "Homologous Recomb.":   ["BRCA1","BRCA2","RAD51","PALB2","RBBP8","EME1","MUS81","BLM","RPA1"],43    "Mismatch Repair":      ["MLH1","MSH2","MSH6","PMS2","PCNA","RFC1","EXO1","POLD1"],44    # Cell Death45    "Apoptosis":            ["TP53","BCL2","BAX","CASP3","CASP9","APAF1","CYCS","BID","PUMA","NOXA"],46    "p53 Signaling":        ["TP53","MDM2","CDKN1A","BAX","PUMA","FAS","GADD45A","SESN2","ZMAT3"],47    # Signaling48    "PI3K/AKT/mTOR":        ["PIK3CA","PTEN","AKT1","MTOR","TSC1","TSC2","FOXO3","S6K1","RICTOR"],49    "RAS/MAPK":             ["KRAS","BRAF","RAF1","MAP2K1","MAPK1","EGFR","GRB2","SOS1","HRAS","NRAS"],50    "WNT/Beta-catenin":     ["APC","CTNNB1","GSK3B","AXIN1","TCF7L2","LRP5","FZD1","DVL1","MYC"],51    "Notch Signaling":      ["NOTCH1","DLL3","JAG1","HES1","RBPJ","MAML1","NUMB","LFNG","PSEN1"],52    "Hedgehog Signaling":   ["SHH","PTCH1","SMO","GLI1","GLI2","SUFU","HHIP","BOC","GAS1"],53    "JAK/STAT":             ["JAK1","JAK2","STAT3","STAT5A","STAT1","IL6","IL2","IFNG","SOCS1","SOCS3"],54    # Immune & Inflammation55    "NF-kB Signaling":      ["NFKB1","RELA","IKBKB","NFKBIA","TNF","IL1B","CXCL8","TRAF2","MAP3K7"],56    "Cytokine Signaling":   ["TNF","IL6","IL1B","CXCL8","CXCL10","CCL2","IFNG","IL10","IL4","IL12A"],57    # Metabolism & Hypoxia58    "Angiogenesis":         ["VEGFA","KDR","FLT1","HIF1A","VHL","PDGFRA","NRP1","ANG","TEK","ANGPT1"],59    "HIF1A / Hypoxia":      ["HIF1A","VHL","EPAS1","ARNT","LDHA","SLC2A1","VEGFA","EPO","BNIP3","CA9"],60    "Oxidative Phosph.":    ["ATP5F1A","NDUFS1","SDHB","UQCRC1","COX5A","TFAM","CYCS","SOD2","CAT"],61    # Structural62    "EMT":                  ["CDH1","VIM","SNAI1","SNAI2","TWIST1","ZEB1","FN1","MMP2","MMP9","TGFB1"],63    "Focal Adhesion":       ["ITGA5","ITGB1","PTK2","SRC","VCL","TLN1","PXN","BCAR1","LIMS1"],64    # Protein Homeostasis65    "Ubiquitin Proteasome": ["PSMD1","UBA1","UBE2C","FBXW7","SKP2","BTRC","VHL","MDM2","HERC2"],66}67 68 69# ─────────────────────────────────────────────70# DEMO DATA71# ─────────────────────────────────────────────72 73def make_demo_de(n=800, seed=42):74    rng = np.random.default_rng(seed)75    known = ["TP53","BRCA1","BRCA2","EGFR","MYC","VEGFA","TNF","IL6","CDKN2A","PTEN",76             "RB1","APC","KRAS","BRAF","PIK3CA","ERBB2","CDH1","VHL","MLH1","ATM",77             "STAT3","JAK2","BCL2","BAX","CASP3","MTOR","AKT1","FOXO3","CDK4","CDK6",78             "CCND1","E2F1","RAD51","PARP1","MSH2","APAF1","CYCS","TSC1","TSC2","NRP1",79             "KDR","FLT1","HIF1A","PDGFRA","CXCL8","NFKB1","IL1B","RAF1","GRB2","MEK1"]80    generic   = [f"GENE_{i:04d}" for i in range(1, n - len(known) + 1)]81    genes     = (known + generic)[:n]82    log2fc    = rng.normal(0, 1.2, n)83    base_mean = rng.lognormal(5, 1.5, n)84    noise     = rng.exponential(1, n)85    pval_raw  = stats.chi2.sf((log2fc**2 / 0.8 + noise * 0.3), df=1)86    rank      = np.argsort(np.argsort(pval_raw)) + 187    pval_adj  = np.clip(pval_raw * n / rank, 0, 1)88    df = pd.DataFrame({89        "gene": genes, "log2FoldChange": log2fc, "pvalue": pval_raw,90        "padj": pval_adj, "baseMean": base_mean,91    })92    return _annotate_de(df)93 94 95def make_demo_counts(de_df, seed=7):96    rng     = np.random.default_rng(seed)97    samples = [f"Ctrl_{i}" for i in range(1,4)] + [f"Treat_{i}" for i in range(1,4)]98    genes   = de_df["gene"].values99    base    = de_df["baseMean"].values100    lfc     = de_df["log2FoldChange"].values101    mat     = np.zeros((len(genes), len(samples)))102    for j, s in enumerate(samples):103        fc    = (2 ** lfc) if "Treat" in s else np.ones(len(genes))104        mat[:,j] = rng.negative_binomial(105            np.clip(base * 0.3, 1, None).astype(int),106            0.3 * np.ones(len(genes)),107        ) * fc108    return pd.DataFrame(np.log2(mat + 1), index=genes, columns=samples)109 110 111# ─────────────────────────────────────────────112# DATA PROCESSING113# ─────────────────────────────────────────────114 115def _annotate_de(df):116    df = df.copy()117    for col in ["padj","pvalue","log2FoldChange"]:118        df[col] = pd.to_numeric(df[col], errors="coerce")119    df["padj"]           = df["padj"].fillna(1.0)120    df["pvalue"]         = df["pvalue"].fillna(1.0)121    df["log2FoldChange"] = df["log2FoldChange"].fillna(0.0)122    df["baseMean"]       = pd.to_numeric(123        df.get("baseMean", pd.Series(np.ones(len(df)) * 100)),124        errors="coerce").fillna(100.0)125    df["-log10padj"] = -np.log10(df["padj"].clip(lower=1e-300))126    df["significant"] = (df["padj"] < 0.05) & (np.abs(df["log2FoldChange"]) > 1)127    df["regulation"]  = "NS"128    df.loc[(df["padj"]<0.05)&(df["log2FoldChange"]> 1),"regulation"] = "Up"129    df.loc[(df["padj"]<0.05)&(df["log2FoldChange"]<-1),"regulation"] = "Down"130    return df131 132 133def _pathway_enrichment(df):134    n_total    = len(df)135    up_genes   = set(df.loc[df["regulation"]=="Up",   "gene"].str.upper())136    down_genes = set(df.loc[df["regulation"]=="Down", "gene"].str.upper())137    bg_up   = max(len(up_genes)   / n_total, 0.001)138    bg_down = max(len(down_genes) / n_total, 0.001)139    rows = []140    for pathway, genes in PATHWAY_DB.items():141        gs        = set(g.upper() for g in genes)142        n_pw      = len(gs)143        hits_up   = len(up_genes   & gs)144        hits_down = len(down_genes & gs)145        score_up   = round((hits_up   / n_pw) / bg_up,   3) if hits_up   > 0 else 0.0146        score_down = round((hits_down / n_pw) / bg_down, 3) if hits_down > 0 else 0.0147        rows.append({148            "pathway": pathway, "n_genes": n_pw,149            "hits_up": hits_up,   "score_up":   score_up,150            "hits_down": hits_down, "score_down": score_down,151            "max_score": max(score_up, score_down),152        })153    return pd.DataFrame(rows).sort_values("max_score", ascending=False)154 155 156# ─────────────────────────────────────────────157# DEMO GLOBALS158# ─────────────────────────────────────────────159DEMO_DE     = make_demo_de()160DEMO_COUNTS = make_demo_counts(DEMO_DE)161 162 163# ─────────────────────────────────────────────164# PLOT BUILDERS165# ─────────────────────────────────────────────166 167def build_volcano(df, fc_thresh=1.0, pval_thresh=0.05, highlight_genes=None):168    df = df.copy()169    df["regulation"] = "NS"170    df.loc[(df["padj"]<pval_thresh)&(df["log2FoldChange"]> fc_thresh),"regulation"] = "Up"171    df.loc[(df["padj"]<pval_thresh)&(df["log2FoldChange"]<-fc_thresh),"regulation"] = "Down"172    cmap = {"Up":"#e63946","Down":"#457b9d","NS":"#adb5bd"}173    smap = {"Up":7,"Down":7,"NS":4}174    omap = {"Up":0.85,"Down":0.85,"NS":0.35}175    fig  = go.Figure()176    for reg in ["NS","Up","Down"]:177        sub = df[df["regulation"]==reg]178        fig.add_trace(go.Scatter(179            x=sub["log2FoldChange"], y=sub["-log10padj"], mode="markers",180            name=f"{reg} ({len(sub)})", text=sub["gene"],181            hovertemplate="<b>%{text}</b><br>log2FC: %{x:.3f}<br>-log10(padj): %{y:.3f}<extra></extra>",182            marker=dict(color=cmap[reg], size=smap[reg], opacity=omap[reg], line=dict(width=0)),183        ))184    if highlight_genes:185        hl = df[df["gene"].isin(highlight_genes)]186        if not hl.empty:187            fig.add_trace(go.Scatter(188                x=hl["log2FoldChange"], y=hl["-log10padj"],189                mode="markers+text", text=hl["gene"], textposition="top center",190                textfont=dict(color="#fbbf24", size=10),191                marker=dict(color="#fbbf24", size=12, symbol="star",192                            line=dict(width=1, color="#fff")),193                name="Highlighted", hoverinfo="skip",194            ))195    fig.add_hline(y=-np.log10(pval_thresh), line_dash="dash", line_color="#6c757d",196                  line_width=1, annotation_text=f"padj={pval_thresh}", annotation_position="right")197    fig.add_vline(x= fc_thresh, line_dash="dash", line_color="#6c757d", line_width=1)198    fig.add_vline(x=-fc_thresh, line_dash="dash", line_color="#6c757d", line_width=1)199    fig.update_layout(200        title=dict(text="Volcano Plot — Differential Expression", font=dict(size=16)),201        xaxis_title="log₂ Fold Change", yaxis_title="-log₁₀ (adjusted p-value)",202        plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),203        legend=dict(bgcolor="rgba(0,0,0,0)", font=dict(size=11)), hovermode="closest",204        xaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),205        yaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),206        margin=dict(l=60,r=20,t=50,b=50),207    )208    return fig209 210 211def build_ma(df):212    df  = df.copy()213    df["log10mean"] = np.log10(df["baseMean"].clip(lower=0.01))214    cmap = {"Up":"#e63946","Down":"#457b9d","NS":"#adb5bd"}215    smap = {"Up":7,"Down":7,"NS":4}216    omap = {"Up":0.85,"Down":0.85,"NS":0.25}217    fig  = go.Figure()218    for reg in ["NS","Up","Down"]:219        sub = df[df["regulation"]==reg]220        fig.add_trace(go.Scatter(221            x=sub["log10mean"], y=sub["log2FoldChange"], mode="markers",222            name=f"{reg} ({len(sub)})", text=sub["gene"],223            hovertemplate="<b>%{text}</b><br>log₁₀(mean): %{x:.3f}<br>log₂FC: %{y:.3f}<extra></extra>",224            marker=dict(color=cmap[reg], size=smap[reg], opacity=omap[reg], line=dict(width=0)),225        ))226    fig.add_hline(y=0,  line_color="#6c757d", line_width=1)227    fig.add_hline(y= 1, line_dash="dash", line_color="#6c757d", line_width=1,228                  annotation_text="FC=2", annotation_position="right")229    fig.add_hline(y=-1, line_dash="dash", line_color="#6c757d", line_width=1)230    fig.update_layout(231        title=dict(text="MA Plot — Mean Expression vs Fold Change", font=dict(size=16)),232        xaxis_title="log₁₀ (mean expression)", yaxis_title="log₂ Fold Change",233        plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),234        legend=dict(bgcolor="rgba(0,0,0,0)", font=dict(size=11)), hovermode="closest",235        xaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),236        yaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),237        margin=dict(l=60,r=20,t=50,b=50),238    )239    return fig240 241 242def build_pca(counts_df):243    mat      = counts_df.values.T244    mat_c    = mat - mat.mean(axis=0)245    U, S, _  = np.linalg.svd(mat_c, full_matrices=False)246    pcs      = U * S247    var      = (S**2) / (S**2).sum() * 100248    samples  = counts_df.columns.tolist()249    groups   = ["Control" if "Ctrl" in s else "Treated" for s in samples]250    fig = go.Figure()251    for grp, color in [("Control","#457b9d"),("Treated","#e63946")]:252        idx = [i for i,g in enumerate(groups) if g==grp]253        fig.add_trace(go.Scatter(254            x=pcs[idx,0], y=pcs[idx,1], mode="markers+text",255            text=[samples[i] for i in idx], textposition="top center",256            textfont=dict(size=9, color="#e2e8f0"),257            marker=dict(size=14, color=color, opacity=0.9,258                        line=dict(width=1.5, color="#fff")),259            name=grp,260            hovertemplate="<b>%{text}</b><br>PC1: %{x:.2f}<br>PC2: %{y:.2f}<extra></extra>",261        ))262    fig.update_layout(263        title=dict(text="PCA — Sample Quality Check", font=dict(size=16)),264        xaxis_title=f"PC1 ({var[0]:.1f}% variance)",265        yaxis_title=f"PC2 ({var[1]:.1f}% variance)",266        plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),267        legend=dict(bgcolor="rgba(0,0,0,0)"),268        xaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),269        yaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),270        margin=dict(l=60,r=20,t=50,b=50),271    )272    return fig, float(var[0]), float(var[1])273 274 275def build_sample_corr(counts_df):276    mat     = counts_df.values.T        # samples × genes277    corr    = np.corrcoef(mat)          # samples × samples278    samples = counts_df.columns.tolist()279    annot   = np.round(corr, 2)280    fig = go.Figure(go.Heatmap(281        z=corr, x=samples, y=samples,282        colorscale="RdBu_r", zmin=-1, zmax=1, zmid=0,283        colorbar=dict(title="Pearson r", tickfont=dict(color="#e2e8f0"),284                      titlefont=dict(color="#e2e8f0")),285        text=annot, texttemplate="%{text}", textfont=dict(size=11),286        hovertemplate="<b>%{x}</b> vs <b>%{y}</b><br>Pearson r: %{z:.4f}<extra></extra>",287    ))288    fig.update_layout(289        title=dict(text="Sample Correlation — Pearson r (log₂ counts)", font=dict(size=16)),290        xaxis=dict(tickangle=-35, tickfont=dict(size=10), side="bottom"),291        yaxis=dict(tickfont=dict(size=10), autorange="reversed"),292        plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),293        margin=dict(l=120,r=20,t=50,b=100), height=520,294    )295    return fig296 297 298def build_heatmap(de_df, counts_df, n_top=40):299    top_genes = de_df.nsmallest(n_top,"padj")["gene"].tolist()300    available = [g for g in top_genes if g in counts_df.index]301    if not available:302        available = counts_df.index[:n_top].tolist()303    sub      = counts_df.loc[available]304    row_mean = sub.mean(axis=1)305    row_std  = sub.std(axis=1).replace(0, 1)306    z        = sub.subtract(row_mean, axis=0).divide(row_std, axis=0)307    if len(z) > 2:308        order = leaves_list(linkage(z.values, method="ward"))309        z     = z.iloc[order]310    fig = go.Figure(go.Heatmap(311        z=z.values, x=z.columns.tolist(), y=z.index.tolist(),312        colorscale="RdBu_r", zmid=0,313        colorbar=dict(title="Z-score", tickfont=dict(color="#e2e8f0"),314                      titlefont=dict(color="#e2e8f0")),315        hovertemplate="Gene: %{y}<br>Sample: %{x}<br>Z-score: %{z:.2f}<extra></extra>",316    ))317    fig.update_layout(318        title=dict(text=f"Heatmap — Top {len(z)} Significant Genes (Z-scored)", font=dict(size=16)),319        xaxis=dict(tickangle=-35, tickfont=dict(size=10)),320        yaxis=dict(tickfont=dict(size=9), autorange="reversed"),321        plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),322        margin=dict(l=120,r=20,t=50,b=80), height=600,323    )324    return fig325 326 327def build_pathway_bar(results):328    top = results.head(20).sort_values("max_score")  # ascending so top is at top of chart329    fig = go.Figure()330    fig.add_trace(go.Bar(331        name="Upregulated genes", y=top["pathway"], x=top["score_up"],332        orientation="h", marker_color="#e63946", opacity=0.85,333        customdata=np.column_stack([top["hits_up"], top["n_genes"]]),334        hovertemplate="<b>%{y}</b><br>Enrichment: %{x:.2f}×<br>Hits: %{customdata[0]}/%{customdata[1]} genes<extra></extra>",335    ))336    fig.add_trace(go.Bar(337        name="Downregulated genes", y=top["pathway"], x=top["score_down"],338        orientation="h", marker_color="#457b9d", opacity=0.85,339        customdata=np.column_stack([top["hits_down"], top["n_genes"]]),340        hovertemplate="<b>%{y}</b><br>Enrichment: %{x:.2f}×<br>Hits: %{customdata[0]}/%{customdata[1]} genes<extra></extra>",341    ))342    fig.update_layout(343        title=dict(text="Pathway Enrichment — Fold Enrichment over Background", font=dict(size=16)),344        xaxis=dict(title="Fold Enrichment (vs background rate)", gridcolor="#1e293b"),345        yaxis=dict(tickfont=dict(size=10)),346        barmode="group",347        plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),348        legend=dict(bgcolor="rgba(0,0,0,0)"),349        height=620, margin=dict(l=190,r=40,t=50,b=60),350    )351    return fig352 353 354def _pathway_table(results):355    disp = results[["pathway","n_genes","hits_up","score_up","hits_down","score_down"]].copy()356    disp.columns = ["Pathway","Pathway Genes","Up Hits","Up Score","Down Hits","Down Score"]357    disp["Up Score"]   = disp["Up Score"].round(2)358    disp["Down Score"] = disp["Down Score"].round(2)359    return dash_table.DataTable(360        data=disp.to_dict("records"),361        columns=[{"name":c,"id":c} for c in disp.columns],362        sort_action="native", page_size=10,363        style_table={"overflowX":"auto","marginTop":"12px"},364        style_cell={"background":CARD,"color":"#e2e8f0","border":"1px solid #334155",365                    "fontSize":"0.8rem","padding":"6px 10px","textAlign":"left"},366        style_header={"background":"#0f172a","fontWeight":"700","color":ACCENT,367                      "border":"1px solid #334155"},368        style_data_conditional=[369            {"if":{"row_index":"odd"},"background":"#162032"},370            {"if":{"filter_query":"{Up Hits} > 0","column_id":"Up Hits"},371             "color":"#e63946","fontWeight":"700"},372            {"if":{"filter_query":"{Down Hits} > 0","column_id":"Down Hits"},373             "color":"#457b9d","fontWeight":"700"},374        ],375    )376 377 378# ─────────────────────────────────────────────379# APP380# ─────────────────────────────────────────────381app = dash.Dash(382    __name__,383    external_stylesheets=[dbc.themes.SLATE, dbc.icons.BOOTSTRAP],384    suppress_callback_exceptions=True,385    title="Transcriptomics Explorer",386)387 388 389def stat_card(icon, value, label, color):390    return dbc.Col(dbc.Card([dbc.CardBody([391        html.Div([392            html.I(className=f"bi bi-{icon} me-2", style={"fontSize":"1.4rem","color":color}),393            html.Span(value, style={"fontSize":"1.6rem","fontWeight":"700","color":color}),394        ], className="d-flex align-items-center"),395        html.P(label, className="mb-0 text-muted small mt-1"),396    ])], style={"background":CARD,"border":f"1px solid {color}22"}), xs=6, md=3)397 398 399app.layout = dbc.Container(fluid=True, style={"background":DARK,"minHeight":"100vh"}, children=[400 401    # HEADER402    dbc.Row(dbc.Col(html.Div([403        html.Div([404            html.I(className="bi bi-activity me-3", style={"fontSize":"2rem","color":ACCENT}),405            html.Div([406                html.H2("Transcriptomics Explorer", className="mb-0",407                        style={"color":ACCENT,"fontWeight":"700"}),408                html.P("Volcano · MA Plot · PCA · Sample Correlation · Heatmap · Pathway Enrichment",409                       className="mb-0 text-muted small"),410            ]),411        ], className="d-flex align-items-center"),412    ], className="py-3 px-2")), className="border-bottom mb-3"),413 414    # UPLOAD + STATS415    dbc.Row([416        dbc.Col([417            dcc.Upload(id="upload-data", multiple=False,418                children=html.Div([419                    html.I(className="bi bi-cloud-upload me-2",420                           style={"fontSize":"1.3rem","color":ACCENT}),421                    html.Span("Drop DESeq2 / edgeR CSV or TSV or "),422                    html.A("Browse", style={"color":ACCENT,"cursor":"pointer"}),423                ]),424                style={"borderWidth":"1px","borderStyle":"dashed","borderRadius":"8px",425                       "borderColor":"#334155","textAlign":"center","padding":"14px",426                       "background":CARD,"color":"#94a3b8","cursor":"pointer"},427            ),428            html.Div("Currently showing: 800 demo genes. Upload your own DESeq2/edgeR results to replace.",429                     className="text-muted small mt-1", id="upload-status"),430        ], md=4),431        dbc.Col(dbc.Row(id="stat-cards"), md=8),432    ], className="mb-3 g-3"),433 434    # TABS435    dbc.Tabs(id="main-tabs", active_tab="tab-volcano", children=[436        dbc.Tab(label="Volcano Plot",        tab_id="tab-volcano",437                label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),438        dbc.Tab(label="MA Plot",             tab_id="tab-ma",439                label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),440        dbc.Tab(label="PCA",                 tab_id="tab-pca",441                label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),442        dbc.Tab(label="Sample Correlation",  tab_id="tab-correlation",443                label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),444        dbc.Tab(label="Heatmap",             tab_id="tab-heatmap",445                label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),446        dbc.Tab(label="Pathway Enrichment",  tab_id="tab-pathway",447                label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),448    ], style={"borderBottom":"1px solid #1e293b"}),449 450    html.Div(id="tab-content", className="mt-3"),451    dcc.Store(id="store-de",     data=DEMO_DE.to_json(orient="split")),452    dcc.Store(id="store-counts", data=DEMO_COUNTS.to_json(orient="split")),453])454 455 456# ─────────────────────────────────────────────457# TAB RENDERER458# ─────────────────────────────────────────────459 460@app.callback(Output("tab-content","children"),461              Input("main-tabs","active_tab"),462              Input("store-de","data"),463              Input("store-counts","data"))464def render_tab(tab, de_json, counts_json):465    df     = pd.read_json(io.StringIO(de_json),     orient="split")466    counts = pd.read_json(io.StringIO(counts_json), orient="split")467 468    if tab == "tab-volcano":469        return dbc.Row([470            dbc.Col(dbc.Card(dbc.CardBody([471                html.H6("Filters", className="text-uppercase mb-3",472                        style={"color":ACCENT,"letterSpacing":"0.08em","fontSize":"0.75rem"}),473                html.Label("log₂FC threshold", className="small text-muted"),474                dcc.Slider(id="fc-slider", min=0, max=4, step=0.1, value=1,475                           marks={i:str(i) for i in range(5)}, tooltip={"placement":"bottom"}),476                html.Label("Adjusted p-value", className="small text-muted mt-3"),477                dcc.Dropdown(id="pval-drop",478                             options=[{"label":f"p < {v}","value":v}479                                      for v in [0.001,0.01,0.05,0.1]],480                             value=0.05, clearable=False,481                             style={"background":"#0f172a","color":"#e2e8f0"}),482                html.Label("Gene search (highlight)", className="small text-muted mt-3"),483                dcc.Input(id="gene-search", type="text", placeholder="e.g. TP53, BRCA1",484                          className="form-control form-control-sm",485                          style={"background":"#0f172a","color":"#e2e8f0","border":"1px solid #334155"}),486                html.Hr(style={"borderColor":"#334155"}),487                html.H6("Summary", className="text-uppercase mb-2",488                        style={"color":ACCENT,"letterSpacing":"0.08em","fontSize":"0.75rem"}),489                html.Div(id="volcano-summary"),490            ]), style={"background":CARD}), md=2),491            dbc.Col([492                dcc.Graph(id="volcano-plot", style={"height":"460px"},493                          config={"displayModeBar":True,"modeBarButtonsToRemove":["lasso2d"]}),494                dbc.Row([495                    dbc.Col(html.H6("Selected / Significant Genes",496                                    className="mt-1 mb-0 text-muted small"), width="auto"),497                    dbc.Col(dbc.Button(498                        [html.I(className="bi bi-download me-1"), "Export CSV"],499                        id="export-btn", size="sm", color="secondary", outline=True,500                    ), width="auto"),501                ], className="mt-3 mb-2 align-items-center"),502                dcc.Download(id="export-download"),503                html.Div(id="de-table-container"),504            ], md=10),505        ], className="g-3")506 507    elif tab == "tab-ma":508        n_up = int((df["regulation"]=="Up").sum())509        n_dn = int((df["regulation"]=="Down").sum())510        return dbc.Row([dbc.Col([511            dcc.Graph(figure=build_ma(df), style={"height":"480px"},512                      config={"displayModeBar":True}),513            dbc.Row([514                dbc.Col(html.Div([html.Span("Upregulated: ", className="text-muted small"),515                    html.Span(str(n_up), style={"color":"#e63946","fontWeight":"700"})]), md=3),516                dbc.Col(html.Div([html.Span("Downregulated: ", className="text-muted small"),517                    html.Span(str(n_dn), style={"color":"#457b9d","fontWeight":"700"})]), md=3),518            ], className="mt-2 px-2"),519            dbc.Alert([html.I(className="bi bi-info-circle me-2"),520                "MA plot shows mean expression (X) vs fold change (Y). "521                "Genes with high expression and large fold change (top/bottom right) "522                "are the most reliable hits. Low-expression genes with high fold change "523                "(top/bottom left) may be noise."],524                color="secondary", className="mt-3 small py-2"),525        ])])526 527    elif tab == "tab-pca":528        fig, var1, var2 = build_pca(counts)529        n_ctrl    = sum("Ctrl"  in c for c in counts.columns)530        n_treated = sum("Treat" in c for c in counts.columns)531        return dbc.Row([dbc.Col([532            dcc.Graph(figure=fig, style={"height":"480px"}, config={"displayModeBar":True}),533            dbc.Row([534                dbc.Col(html.Div([html.Span("PC1 variance: ", className="text-muted small"),535                    html.Span(f"{var1:.1f}%", style={"color":ACCENT,"fontWeight":"700"})]), md=3),536                dbc.Col(html.Div([html.Span("PC2 variance: ", className="text-muted small"),537                    html.Span(f"{var2:.1f}%", style={"color":ACCENT,"fontWeight":"700"})]), md=3),538                dbc.Col(html.Div([html.Span("Control: ", className="text-muted small"),539                    html.Span(str(n_ctrl), style={"color":"#457b9d","fontWeight":"700"})]), md=3),540                dbc.Col(html.Div([html.Span("Treated: ", className="text-muted small"),541                    html.Span(str(n_treated), style={"color":"#e63946","fontWeight":"700"})]), md=3),542            ], className="mt-2 px-2"),543            dbc.Alert([html.I(className="bi bi-info-circle me-2"),544                "Good QC: Control samples cluster together (blue), Treated samples cluster "545                "together (red), with clear separation between groups. Outlier samples "546                "that don't cluster with their group should be investigated. "547                "PCA here is simulated from DE results — upload a raw count matrix for real PCA."],548                color="secondary", className="mt-3 small py-2"),549        ])])550 551    elif tab == "tab-correlation":552        return dbc.Row([dbc.Col([553            dcc.Graph(figure=build_sample_corr(counts), style={"height":"540px"},554                      config={"displayModeBar":True}),555            dbc.Alert([html.I(className="bi bi-info-circle me-2"),556                "Values close to 1.0 = highly similar samples. "557                "Replicates within the same group (e.g. Ctrl_1 vs Ctrl_2) should have "558                "r > 0.95. Low correlation within a group may indicate a failed sample. "559                "Cross-group correlation (Control vs Treated) is expected to be lower."],560                color="secondary", className="mt-2 small py-2"),561        ])])562 563    elif tab == "tab-heatmap":564        n_top = min(40, max(int(df["significant"].sum()), 20))565        return dbc.Row([dbc.Col([566            dcc.Graph(figure=build_heatmap(df, counts, n_top=n_top),567                      style={"height":"650px"}, config={"displayModeBar":True}),568            dbc.Alert([html.I(className="bi bi-info-circle me-2"),569                "Top significant genes by adjusted p-value. "570                "Red = high expression, Blue = low (Z-scored per gene). "571                "Genes are clustered by expression similarity — "572                "genes in the same cluster behave similarly across samples."],573                color="secondary", className="mt-2 small py-2"),574        ])])575 576    elif tab == "tab-pathway":577        results   = _pathway_enrichment(df)578        sig_count = int(df["significant"].sum())579        n_enriched = int((results["max_score"] > 0).sum())580        return dbc.Row([dbc.Col([581            dcc.Graph(figure=build_pathway_bar(results), style={"height":"650px"},582                      config={"displayModeBar":True}),583            dbc.Row([584                dbc.Col(html.Div([html.Span("Significant genes tested: ", className="text-muted small"),585                    html.Span(str(sig_count), style={"color":ACCENT,"fontWeight":"700"})]), md=4),586                dbc.Col(html.Div([html.Span("Pathways with hits: ", className="text-muted small"),587                    html.Span(str(n_enriched), style={"color":"#2ecc71","fontWeight":"700"})]), md=4),588            ], className="mt-2 px-2"),589            html.H6("Enrichment Summary", className="mt-3 mb-1 text-muted small"),590            _pathway_table(results),591            dbc.Alert([html.I(className="bi bi-info-circle me-2"),592                "Enrichment score = fraction of pathway genes that are significant in your data "593                "divided by the background rate. Score > 1 means the pathway is enriched. "594                "Uses 22 representative KEGG/Reactome pathways. "595                "For full pathway analysis, run clusterProfiler in R after DESeq2."],596                color="secondary", className="mt-2 small py-2"),597        ])])598 599 600# ─────────────────────────────────────────────601# CALLBACKS602# ─────────────────────────────────────────────603 604@app.callback(Output("stat-cards","children"), Input("store-de","data"))605def update_stats(de_json):606    df = pd.read_json(io.StringIO(de_json), orient="split")607    return [608        stat_card("arrow-up-circle",   int((df["regulation"]=="Up").sum()),   "Upregulated",         "#e63946"),609        stat_card("arrow-down-circle", int((df["regulation"]=="Down").sum()), "Downregulated",       "#457b9d"),610        stat_card("check2-circle",     int(df["significant"].sum()),          "Significant (p<0.05)","#2ecc71"),611        stat_card("bar-chart",         f"{len(df):,}",                        "Total Genes",         ACCENT),612    ]613 614 615@app.callback(616    Output("store-de","data"),617    Output("store-counts","data"),618    Output("upload-status","children"),619    Input("upload-data","contents"),620    State("upload-data","filename"),621    prevent_initial_call=True,622)623def handle_upload(contents, filename):624    if not contents:625        return dash.no_update, dash.no_update, ""626    _, content_string = contents.split(",")627    decoded = base64.b64decode(content_string)628    try:629        sep = "\t" if (filename or "").endswith(".tsv") else ","630        df  = pd.read_csv(io.StringIO(decoded.decode("utf-8")), sep=sep)631        rename = {}632        for col in df.columns:633            lc = col.lower()634            if   lc in ("gene","gene_id","geneid","symbol"):       rename[col]="gene"635            elif lc in ("log2foldchange","log2fc","lfc"):          rename[col]="log2FoldChange"636            elif lc in ("pvalue","pval","p.value","p_value"):      rename[col]="pvalue"637            elif lc in ("padj","p.adj","fdr","adjusted_pvalue"):   rename[col]="padj"638            elif lc in ("basemean","mean_expr","avgexpr"):         rename[col]="baseMean"639        df = df.rename(columns=rename)640        missing = {"gene","log2FoldChange","pvalue","padj"} - set(df.columns)641        if missing:642            return dash.no_update, dash.no_update, dbc.Alert(643                f"Missing columns: {missing}", color="danger", className="py-1")644        df     = _annotate_de(df)645        counts = make_demo_counts(df)646        return df.to_json(orient="split"), counts.to_json(orient="split"), dbc.Alert(647            f"Loaded {filename}: {len(df):,} genes", color="success", className="py-1")648    except Exception as e:649        return dash.no_update, dash.no_update, dbc.Alert(650            f"Error: {e}", color="danger", className="py-1")651 652 653@app.callback(654    Output("volcano-plot","figure"),655    Output("volcano-summary","children"),656    Input("fc-slider","value"),657    Input("pval-drop","value"),658    Input("gene-search","value"),659    Input("store-de","data"),660)661def update_volcano(fc, pval, search, de_json):662    df         = pd.read_json(io.StringIO(de_json), orient="split")663    highlights = [g.strip() for g in (search or "").split(",") if g.strip()]664    fig        = build_volcano(df, fc_thresh=fc, pval_thresh=pval, highlight_genes=highlights)665    n_up = int(((df["padj"]<pval)&(df["log2FoldChange"]> fc)).sum())666    n_dn = int(((df["padj"]<pval)&(df["log2FoldChange"]<-fc)).sum())667    summary = [668        html.Div([html.Span("▲ Up: ",   style={"color":"#e63946"}), html.Span(str(n_up))], className="small mb-1"),669        html.Div([html.Span("▼ Down: ", style={"color":"#457b9d"}), html.Span(str(n_dn))], className="small"),670    ]671    return fig, summary672 673 674@app.callback(675    Output("de-table-container","children"),676    Input("volcano-plot","selectedData"),677    Input("fc-slider","value"),678    Input("pval-drop","value"),679    Input("store-de","data"),680)681def update_de_table(selected, fc, pval, de_json):682    df = pd.read_json(io.StringIO(de_json), orient="split")683    df["regulation"] = "NS"684    df.loc[(df["padj"]<pval)&(df["log2FoldChange"]> fc),"regulation"] = "Up"685    df.loc[(df["padj"]<pval)&(df["log2FoldChange"]<-fc),"regulation"] = "Down"686    if selected and selected.get("points"):687        sub = df[df["gene"].isin([p["text"] for p in selected["points"] if "text" in p])]688    else:689        sub = df[df["regulation"]!="NS"].nlargest(50,"-log10padj")690    sub = sub[["gene","log2FoldChange","pvalue","padj","baseMean","regulation"]].copy()691    sub["log2FoldChange"] = sub["log2FoldChange"].round(3)692    sub["pvalue"]   = sub["pvalue"].apply(lambda x: f"{x:.2e}")693    sub["padj"]     = sub["padj"].apply(lambda x: f"{x:.2e}")694    sub["baseMean"] = sub["baseMean"].round(1)695    return dash_table.DataTable(696        data=sub.to_dict("records"),697        columns=[{"name":c,"id":c,"type":"text"} for c in sub.columns],698        filter_action="native", sort_action="native", page_size=12,699        style_table={"overflowX":"auto"},700        style_cell={"background":CARD,"color":"#e2e8f0","border":"1px solid #334155",701                    "fontSize":"0.8rem","padding":"6px 10px","textAlign":"left"},702        style_header={"background":"#0f172a","fontWeight":"700","color":ACCENT,703                      "border":"1px solid #334155"},704        style_data_conditional=[705            {"if":{"filter_query":'{regulation} = "Up"',   "column_id":"regulation"},"color":"#e63946","fontWeight":"700"},706            {"if":{"filter_query":'{regulation} = "Down"', "column_id":"regulation"},"color":"#457b9d","fontWeight":"700"},707            {"if":{"row_index":"odd"},"background":"#162032"},708        ],709        style_filter={"background":"#0f172a","color":"#e2e8f0","border":"1px solid #334155"},710    )711 712 713@app.callback(714    Output("export-download","data"),715    Input("export-btn","n_clicks"),716    State("store-de","data"),717    State("fc-slider","value"),718    State("pval-drop","value"),719    prevent_initial_call=True,720)721def export_csv(_, de_json, fc, pval):722    df = pd.read_json(io.StringIO(de_json), orient="split")723    df["regulation"] = "NS"724    df.loc[(df["padj"]<pval)&(df["log2FoldChange"]> fc),"regulation"] = "Up"725    df.loc[(df["padj"]<pval)&(df["log2FoldChange"]<-fc),"regulation"] = "Down"726    sig = df[df["regulation"]!="NS"].sort_values("-log10padj", ascending=False)727    return dcc.send_data_frame(728        sig[["gene","log2FoldChange","pvalue","padj","baseMean","regulation"]].to_csv,729        "significant_genes.csv", index=False,730    )731 732 733# ─────────────────────────────────────────────734if __name__ == "__main__":735    import os736    port = int(os.environ.get("PORT", 7860))737    app.run(debug=False, host="0.0.0.0", port=port)738