JacksonFW/transcriptomics-explorer
0
1"""2Transcriptomics Explorer3Interactive RNA-seq / differential expression dashboard:4 - Volcano Plot — significant DE genes5 - MA Plot — mean expression vs fold change6 - PCA — sample quality check7 - Sample Corr. — sample-to-sample Pearson correlation8 - Heatmap — top gene expression patterns9 - Pathway Enrich. — enriched biological pathways10 11Run: python app.py12Open: http://localhost:805013"""14 15import base6416import io17 18import dash19import dash_bootstrap_components as dbc20import numpy as np21import pandas as pd22import plotly.express as px23import plotly.graph_objects as go24from dash import Input, Output, State, dash_table, dcc, html25from scipy import stats26from scipy.cluster.hierarchy import leaves_list, linkage27 28# ─────────────────────────────────────────────29# CONSTANTS30# ─────────────────────────────────────────────31DARK = "#0f172a"32CARD = "#1e293b"33ACCENT = "#38bdf8"34 35PATHWAY_DB = {36 # Cell Proliferation & Cycle37 "Cell Cycle": ["TP53","RB1","CDKN2A","CDK4","CDK6","E2F1","CCND1","CDC20","PLK1","AURKA"],38 "G1/S Transition": ["CCND1","CDK4","CDK6","RB1","E2F1","CDKN1A","CDKN2A","CCNE1","CDK2"],39 "DNA Replication": ["PCNA","MCM2","MCM7","RFC1","POLA1","POLE","RPA1","GINS1","CDC6"],40 # DNA Damage & Repair41 "DNA Repair": ["BRCA1","BRCA2","ATM","MLH1","MSH2","RAD51","PARP1","ATR","CHEK1","CHEK2"],42 "Homologous Recomb.": ["BRCA1","BRCA2","RAD51","PALB2","RBBP8","EME1","MUS81","BLM","RPA1"],43 "Mismatch Repair": ["MLH1","MSH2","MSH6","PMS2","PCNA","RFC1","EXO1","POLD1"],44 # Cell Death45 "Apoptosis": ["TP53","BCL2","BAX","CASP3","CASP9","APAF1","CYCS","BID","PUMA","NOXA"],46 "p53 Signaling": ["TP53","MDM2","CDKN1A","BAX","PUMA","FAS","GADD45A","SESN2","ZMAT3"],47 # Signaling48 "PI3K/AKT/mTOR": ["PIK3CA","PTEN","AKT1","MTOR","TSC1","TSC2","FOXO3","S6K1","RICTOR"],49 "RAS/MAPK": ["KRAS","BRAF","RAF1","MAP2K1","MAPK1","EGFR","GRB2","SOS1","HRAS","NRAS"],50 "WNT/Beta-catenin": ["APC","CTNNB1","GSK3B","AXIN1","TCF7L2","LRP5","FZD1","DVL1","MYC"],51 "Notch Signaling": ["NOTCH1","DLL3","JAG1","HES1","RBPJ","MAML1","NUMB","LFNG","PSEN1"],52 "Hedgehog Signaling": ["SHH","PTCH1","SMO","GLI1","GLI2","SUFU","HHIP","BOC","GAS1"],53 "JAK/STAT": ["JAK1","JAK2","STAT3","STAT5A","STAT1","IL6","IL2","IFNG","SOCS1","SOCS3"],54 # Immune & Inflammation55 "NF-kB Signaling": ["NFKB1","RELA","IKBKB","NFKBIA","TNF","IL1B","CXCL8","TRAF2","MAP3K7"],56 "Cytokine Signaling": ["TNF","IL6","IL1B","CXCL8","CXCL10","CCL2","IFNG","IL10","IL4","IL12A"],57 # Metabolism & Hypoxia58 "Angiogenesis": ["VEGFA","KDR","FLT1","HIF1A","VHL","PDGFRA","NRP1","ANG","TEK","ANGPT1"],59 "HIF1A / Hypoxia": ["HIF1A","VHL","EPAS1","ARNT","LDHA","SLC2A1","VEGFA","EPO","BNIP3","CA9"],60 "Oxidative Phosph.": ["ATP5F1A","NDUFS1","SDHB","UQCRC1","COX5A","TFAM","CYCS","SOD2","CAT"],61 # Structural62 "EMT": ["CDH1","VIM","SNAI1","SNAI2","TWIST1","ZEB1","FN1","MMP2","MMP9","TGFB1"],63 "Focal Adhesion": ["ITGA5","ITGB1","PTK2","SRC","VCL","TLN1","PXN","BCAR1","LIMS1"],64 # Protein Homeostasis65 "Ubiquitin Proteasome": ["PSMD1","UBA1","UBE2C","FBXW7","SKP2","BTRC","VHL","MDM2","HERC2"],66}67 68 69# ─────────────────────────────────────────────70# DEMO DATA71# ─────────────────────────────────────────────72 73def make_demo_de(n=800, seed=42):74 rng = np.random.default_rng(seed)75 known = ["TP53","BRCA1","BRCA2","EGFR","MYC","VEGFA","TNF","IL6","CDKN2A","PTEN",76 "RB1","APC","KRAS","BRAF","PIK3CA","ERBB2","CDH1","VHL","MLH1","ATM",77 "STAT3","JAK2","BCL2","BAX","CASP3","MTOR","AKT1","FOXO3","CDK4","CDK6",78 "CCND1","E2F1","RAD51","PARP1","MSH2","APAF1","CYCS","TSC1","TSC2","NRP1",79 "KDR","FLT1","HIF1A","PDGFRA","CXCL8","NFKB1","IL1B","RAF1","GRB2","MEK1"]80 generic = [f"GENE_{i:04d}" for i in range(1, n - len(known) + 1)]81 genes = (known + generic)[:n]82 log2fc = rng.normal(0, 1.2, n)83 base_mean = rng.lognormal(5, 1.5, n)84 noise = rng.exponential(1, n)85 pval_raw = stats.chi2.sf((log2fc**2 / 0.8 + noise * 0.3), df=1)86 rank = np.argsort(np.argsort(pval_raw)) + 187 pval_adj = np.clip(pval_raw * n / rank, 0, 1)88 df = pd.DataFrame({89 "gene": genes, "log2FoldChange": log2fc, "pvalue": pval_raw,90 "padj": pval_adj, "baseMean": base_mean,91 })92 return _annotate_de(df)93 94 95def make_demo_counts(de_df, seed=7):96 rng = np.random.default_rng(seed)97 samples = [f"Ctrl_{i}" for i in range(1,4)] + [f"Treat_{i}" for i in range(1,4)]98 genes = de_df["gene"].values99 base = de_df["baseMean"].values100 lfc = de_df["log2FoldChange"].values101 mat = np.zeros((len(genes), len(samples)))102 for j, s in enumerate(samples):103 fc = (2 ** lfc) if "Treat" in s else np.ones(len(genes))104 mat[:,j] = rng.negative_binomial(105 np.clip(base * 0.3, 1, None).astype(int),106 0.3 * np.ones(len(genes)),107 ) * fc108 return pd.DataFrame(np.log2(mat + 1), index=genes, columns=samples)109 110 111# ─────────────────────────────────────────────112# DATA PROCESSING113# ─────────────────────────────────────────────114 115def _annotate_de(df):116 df = df.copy()117 for col in ["padj","pvalue","log2FoldChange"]:118 df[col] = pd.to_numeric(df[col], errors="coerce")119 df["padj"] = df["padj"].fillna(1.0)120 df["pvalue"] = df["pvalue"].fillna(1.0)121 df["log2FoldChange"] = df["log2FoldChange"].fillna(0.0)122 df["baseMean"] = pd.to_numeric(123 df.get("baseMean", pd.Series(np.ones(len(df)) * 100)),124 errors="coerce").fillna(100.0)125 df["-log10padj"] = -np.log10(df["padj"].clip(lower=1e-300))126 df["significant"] = (df["padj"] < 0.05) & (np.abs(df["log2FoldChange"]) > 1)127 df["regulation"] = "NS"128 df.loc[(df["padj"]<0.05)&(df["log2FoldChange"]> 1),"regulation"] = "Up"129 df.loc[(df["padj"]<0.05)&(df["log2FoldChange"]<-1),"regulation"] = "Down"130 return df131 132 133def _pathway_enrichment(df):134 n_total = len(df)135 up_genes = set(df.loc[df["regulation"]=="Up", "gene"].str.upper())136 down_genes = set(df.loc[df["regulation"]=="Down", "gene"].str.upper())137 bg_up = max(len(up_genes) / n_total, 0.001)138 bg_down = max(len(down_genes) / n_total, 0.001)139 rows = []140 for pathway, genes in PATHWAY_DB.items():141 gs = set(g.upper() for g in genes)142 n_pw = len(gs)143 hits_up = len(up_genes & gs)144 hits_down = len(down_genes & gs)145 score_up = round((hits_up / n_pw) / bg_up, 3) if hits_up > 0 else 0.0146 score_down = round((hits_down / n_pw) / bg_down, 3) if hits_down > 0 else 0.0147 rows.append({148 "pathway": pathway, "n_genes": n_pw,149 "hits_up": hits_up, "score_up": score_up,150 "hits_down": hits_down, "score_down": score_down,151 "max_score": max(score_up, score_down),152 })153 return pd.DataFrame(rows).sort_values("max_score", ascending=False)154 155 156# ─────────────────────────────────────────────157# DEMO GLOBALS158# ─────────────────────────────────────────────159DEMO_DE = make_demo_de()160DEMO_COUNTS = make_demo_counts(DEMO_DE)161 162 163# ─────────────────────────────────────────────164# PLOT BUILDERS165# ─────────────────────────────────────────────166 167def build_volcano(df, fc_thresh=1.0, pval_thresh=0.05, highlight_genes=None):168 df = df.copy()169 df["regulation"] = "NS"170 df.loc[(df["padj"]<pval_thresh)&(df["log2FoldChange"]> fc_thresh),"regulation"] = "Up"171 df.loc[(df["padj"]<pval_thresh)&(df["log2FoldChange"]<-fc_thresh),"regulation"] = "Down"172 cmap = {"Up":"#e63946","Down":"#457b9d","NS":"#adb5bd"}173 smap = {"Up":7,"Down":7,"NS":4}174 omap = {"Up":0.85,"Down":0.85,"NS":0.35}175 fig = go.Figure()176 for reg in ["NS","Up","Down"]:177 sub = df[df["regulation"]==reg]178 fig.add_trace(go.Scatter(179 x=sub["log2FoldChange"], y=sub["-log10padj"], mode="markers",180 name=f"{reg} ({len(sub)})", text=sub["gene"],181 hovertemplate="<b>%{text}</b><br>log2FC: %{x:.3f}<br>-log10(padj): %{y:.3f}<extra></extra>",182 marker=dict(color=cmap[reg], size=smap[reg], opacity=omap[reg], line=dict(width=0)),183 ))184 if highlight_genes:185 hl = df[df["gene"].isin(highlight_genes)]186 if not hl.empty:187 fig.add_trace(go.Scatter(188 x=hl["log2FoldChange"], y=hl["-log10padj"],189 mode="markers+text", text=hl["gene"], textposition="top center",190 textfont=dict(color="#fbbf24", size=10),191 marker=dict(color="#fbbf24", size=12, symbol="star",192 line=dict(width=1, color="#fff")),193 name="Highlighted", hoverinfo="skip",194 ))195 fig.add_hline(y=-np.log10(pval_thresh), line_dash="dash", line_color="#6c757d",196 line_width=1, annotation_text=f"padj={pval_thresh}", annotation_position="right")197 fig.add_vline(x= fc_thresh, line_dash="dash", line_color="#6c757d", line_width=1)198 fig.add_vline(x=-fc_thresh, line_dash="dash", line_color="#6c757d", line_width=1)199 fig.update_layout(200 title=dict(text="Volcano Plot — Differential Expression", font=dict(size=16)),201 xaxis_title="log₂ Fold Change", yaxis_title="-log₁₀ (adjusted p-value)",202 plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),203 legend=dict(bgcolor="rgba(0,0,0,0)", font=dict(size=11)), hovermode="closest",204 xaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),205 yaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),206 margin=dict(l=60,r=20,t=50,b=50),207 )208 return fig209 210 211def build_ma(df):212 df = df.copy()213 df["log10mean"] = np.log10(df["baseMean"].clip(lower=0.01))214 cmap = {"Up":"#e63946","Down":"#457b9d","NS":"#adb5bd"}215 smap = {"Up":7,"Down":7,"NS":4}216 omap = {"Up":0.85,"Down":0.85,"NS":0.25}217 fig = go.Figure()218 for reg in ["NS","Up","Down"]:219 sub = df[df["regulation"]==reg]220 fig.add_trace(go.Scatter(221 x=sub["log10mean"], y=sub["log2FoldChange"], mode="markers",222 name=f"{reg} ({len(sub)})", text=sub["gene"],223 hovertemplate="<b>%{text}</b><br>log₁₀(mean): %{x:.3f}<br>log₂FC: %{y:.3f}<extra></extra>",224 marker=dict(color=cmap[reg], size=smap[reg], opacity=omap[reg], line=dict(width=0)),225 ))226 fig.add_hline(y=0, line_color="#6c757d", line_width=1)227 fig.add_hline(y= 1, line_dash="dash", line_color="#6c757d", line_width=1,228 annotation_text="FC=2", annotation_position="right")229 fig.add_hline(y=-1, line_dash="dash", line_color="#6c757d", line_width=1)230 fig.update_layout(231 title=dict(text="MA Plot — Mean Expression vs Fold Change", font=dict(size=16)),232 xaxis_title="log₁₀ (mean expression)", yaxis_title="log₂ Fold Change",233 plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),234 legend=dict(bgcolor="rgba(0,0,0,0)", font=dict(size=11)), hovermode="closest",235 xaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),236 yaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),237 margin=dict(l=60,r=20,t=50,b=50),238 )239 return fig240 241 242def build_pca(counts_df):243 mat = counts_df.values.T244 mat_c = mat - mat.mean(axis=0)245 U, S, _ = np.linalg.svd(mat_c, full_matrices=False)246 pcs = U * S247 var = (S**2) / (S**2).sum() * 100248 samples = counts_df.columns.tolist()249 groups = ["Control" if "Ctrl" in s else "Treated" for s in samples]250 fig = go.Figure()251 for grp, color in [("Control","#457b9d"),("Treated","#e63946")]:252 idx = [i for i,g in enumerate(groups) if g==grp]253 fig.add_trace(go.Scatter(254 x=pcs[idx,0], y=pcs[idx,1], mode="markers+text",255 text=[samples[i] for i in idx], textposition="top center",256 textfont=dict(size=9, color="#e2e8f0"),257 marker=dict(size=14, color=color, opacity=0.9,258 line=dict(width=1.5, color="#fff")),259 name=grp,260 hovertemplate="<b>%{text}</b><br>PC1: %{x:.2f}<br>PC2: %{y:.2f}<extra></extra>",261 ))262 fig.update_layout(263 title=dict(text="PCA — Sample Quality Check", font=dict(size=16)),264 xaxis_title=f"PC1 ({var[0]:.1f}% variance)",265 yaxis_title=f"PC2 ({var[1]:.1f}% variance)",266 plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),267 legend=dict(bgcolor="rgba(0,0,0,0)"),268 xaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),269 yaxis=dict(gridcolor="#1e293b", zerolinecolor="#334155"),270 margin=dict(l=60,r=20,t=50,b=50),271 )272 return fig, float(var[0]), float(var[1])273 274 275def build_sample_corr(counts_df):276 mat = counts_df.values.T # samples × genes277 corr = np.corrcoef(mat) # samples × samples278 samples = counts_df.columns.tolist()279 annot = np.round(corr, 2)280 fig = go.Figure(go.Heatmap(281 z=corr, x=samples, y=samples,282 colorscale="RdBu_r", zmin=-1, zmax=1, zmid=0,283 colorbar=dict(title="Pearson r", tickfont=dict(color="#e2e8f0"),284 titlefont=dict(color="#e2e8f0")),285 text=annot, texttemplate="%{text}", textfont=dict(size=11),286 hovertemplate="<b>%{x}</b> vs <b>%{y}</b><br>Pearson r: %{z:.4f}<extra></extra>",287 ))288 fig.update_layout(289 title=dict(text="Sample Correlation — Pearson r (log₂ counts)", font=dict(size=16)),290 xaxis=dict(tickangle=-35, tickfont=dict(size=10), side="bottom"),291 yaxis=dict(tickfont=dict(size=10), autorange="reversed"),292 plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),293 margin=dict(l=120,r=20,t=50,b=100), height=520,294 )295 return fig296 297 298def build_heatmap(de_df, counts_df, n_top=40):299 top_genes = de_df.nsmallest(n_top,"padj")["gene"].tolist()300 available = [g for g in top_genes if g in counts_df.index]301 if not available:302 available = counts_df.index[:n_top].tolist()303 sub = counts_df.loc[available]304 row_mean = sub.mean(axis=1)305 row_std = sub.std(axis=1).replace(0, 1)306 z = sub.subtract(row_mean, axis=0).divide(row_std, axis=0)307 if len(z) > 2:308 order = leaves_list(linkage(z.values, method="ward"))309 z = z.iloc[order]310 fig = go.Figure(go.Heatmap(311 z=z.values, x=z.columns.tolist(), y=z.index.tolist(),312 colorscale="RdBu_r", zmid=0,313 colorbar=dict(title="Z-score", tickfont=dict(color="#e2e8f0"),314 titlefont=dict(color="#e2e8f0")),315 hovertemplate="Gene: %{y}<br>Sample: %{x}<br>Z-score: %{z:.2f}<extra></extra>",316 ))317 fig.update_layout(318 title=dict(text=f"Heatmap — Top {len(z)} Significant Genes (Z-scored)", font=dict(size=16)),319 xaxis=dict(tickangle=-35, tickfont=dict(size=10)),320 yaxis=dict(tickfont=dict(size=9), autorange="reversed"),321 plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),322 margin=dict(l=120,r=20,t=50,b=80), height=600,323 )324 return fig325 326 327def build_pathway_bar(results):328 top = results.head(20).sort_values("max_score") # ascending so top is at top of chart329 fig = go.Figure()330 fig.add_trace(go.Bar(331 name="Upregulated genes", y=top["pathway"], x=top["score_up"],332 orientation="h", marker_color="#e63946", opacity=0.85,333 customdata=np.column_stack([top["hits_up"], top["n_genes"]]),334 hovertemplate="<b>%{y}</b><br>Enrichment: %{x:.2f}×<br>Hits: %{customdata[0]}/%{customdata[1]} genes<extra></extra>",335 ))336 fig.add_trace(go.Bar(337 name="Downregulated genes", y=top["pathway"], x=top["score_down"],338 orientation="h", marker_color="#457b9d", opacity=0.85,339 customdata=np.column_stack([top["hits_down"], top["n_genes"]]),340 hovertemplate="<b>%{y}</b><br>Enrichment: %{x:.2f}×<br>Hits: %{customdata[0]}/%{customdata[1]} genes<extra></extra>",341 ))342 fig.update_layout(343 title=dict(text="Pathway Enrichment — Fold Enrichment over Background", font=dict(size=16)),344 xaxis=dict(title="Fold Enrichment (vs background rate)", gridcolor="#1e293b"),345 yaxis=dict(tickfont=dict(size=10)),346 barmode="group",347 plot_bgcolor=DARK, paper_bgcolor="rgba(0,0,0,0)", font=dict(color="#e2e8f0"),348 legend=dict(bgcolor="rgba(0,0,0,0)"),349 height=620, margin=dict(l=190,r=40,t=50,b=60),350 )351 return fig352 353 354def _pathway_table(results):355 disp = results[["pathway","n_genes","hits_up","score_up","hits_down","score_down"]].copy()356 disp.columns = ["Pathway","Pathway Genes","Up Hits","Up Score","Down Hits","Down Score"]357 disp["Up Score"] = disp["Up Score"].round(2)358 disp["Down Score"] = disp["Down Score"].round(2)359 return dash_table.DataTable(360 data=disp.to_dict("records"),361 columns=[{"name":c,"id":c} for c in disp.columns],362 sort_action="native", page_size=10,363 style_table={"overflowX":"auto","marginTop":"12px"},364 style_cell={"background":CARD,"color":"#e2e8f0","border":"1px solid #334155",365 "fontSize":"0.8rem","padding":"6px 10px","textAlign":"left"},366 style_header={"background":"#0f172a","fontWeight":"700","color":ACCENT,367 "border":"1px solid #334155"},368 style_data_conditional=[369 {"if":{"row_index":"odd"},"background":"#162032"},370 {"if":{"filter_query":"{Up Hits} > 0","column_id":"Up Hits"},371 "color":"#e63946","fontWeight":"700"},372 {"if":{"filter_query":"{Down Hits} > 0","column_id":"Down Hits"},373 "color":"#457b9d","fontWeight":"700"},374 ],375 )376 377 378# ─────────────────────────────────────────────379# APP380# ─────────────────────────────────────────────381app = dash.Dash(382 __name__,383 external_stylesheets=[dbc.themes.SLATE, dbc.icons.BOOTSTRAP],384 suppress_callback_exceptions=True,385 title="Transcriptomics Explorer",386)387 388 389def stat_card(icon, value, label, color):390 return dbc.Col(dbc.Card([dbc.CardBody([391 html.Div([392 html.I(className=f"bi bi-{icon} me-2", style={"fontSize":"1.4rem","color":color}),393 html.Span(value, style={"fontSize":"1.6rem","fontWeight":"700","color":color}),394 ], className="d-flex align-items-center"),395 html.P(label, className="mb-0 text-muted small mt-1"),396 ])], style={"background":CARD,"border":f"1px solid {color}22"}), xs=6, md=3)397 398 399app.layout = dbc.Container(fluid=True, style={"background":DARK,"minHeight":"100vh"}, children=[400 401 # HEADER402 dbc.Row(dbc.Col(html.Div([403 html.Div([404 html.I(className="bi bi-activity me-3", style={"fontSize":"2rem","color":ACCENT}),405 html.Div([406 html.H2("Transcriptomics Explorer", className="mb-0",407 style={"color":ACCENT,"fontWeight":"700"}),408 html.P("Volcano · MA Plot · PCA · Sample Correlation · Heatmap · Pathway Enrichment",409 className="mb-0 text-muted small"),410 ]),411 ], className="d-flex align-items-center"),412 ], className="py-3 px-2")), className="border-bottom mb-3"),413 414 # UPLOAD + STATS415 dbc.Row([416 dbc.Col([417 dcc.Upload(id="upload-data", multiple=False,418 children=html.Div([419 html.I(className="bi bi-cloud-upload me-2",420 style={"fontSize":"1.3rem","color":ACCENT}),421 html.Span("Drop DESeq2 / edgeR CSV or TSV or "),422 html.A("Browse", style={"color":ACCENT,"cursor":"pointer"}),423 ]),424 style={"borderWidth":"1px","borderStyle":"dashed","borderRadius":"8px",425 "borderColor":"#334155","textAlign":"center","padding":"14px",426 "background":CARD,"color":"#94a3b8","cursor":"pointer"},427 ),428 html.Div("Currently showing: 800 demo genes. Upload your own DESeq2/edgeR results to replace.",429 className="text-muted small mt-1", id="upload-status"),430 ], md=4),431 dbc.Col(dbc.Row(id="stat-cards"), md=8),432 ], className="mb-3 g-3"),433 434 # TABS435 dbc.Tabs(id="main-tabs", active_tab="tab-volcano", children=[436 dbc.Tab(label="Volcano Plot", tab_id="tab-volcano",437 label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),438 dbc.Tab(label="MA Plot", tab_id="tab-ma",439 label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),440 dbc.Tab(label="PCA", tab_id="tab-pca",441 label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),442 dbc.Tab(label="Sample Correlation", tab_id="tab-correlation",443 label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),444 dbc.Tab(label="Heatmap", tab_id="tab-heatmap",445 label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),446 dbc.Tab(label="Pathway Enrichment", tab_id="tab-pathway",447 label_style={"color":"#94a3b8"}, active_label_style={"color":ACCENT}),448 ], style={"borderBottom":"1px solid #1e293b"}),449 450 html.Div(id="tab-content", className="mt-3"),451 dcc.Store(id="store-de", data=DEMO_DE.to_json(orient="split")),452 dcc.Store(id="store-counts", data=DEMO_COUNTS.to_json(orient="split")),453])454 455 456# ─────────────────────────────────────────────457# TAB RENDERER458# ─────────────────────────────────────────────459 460@app.callback(Output("tab-content","children"),461 Input("main-tabs","active_tab"),462 Input("store-de","data"),463 Input("store-counts","data"))464def render_tab(tab, de_json, counts_json):465 df = pd.read_json(io.StringIO(de_json), orient="split")466 counts = pd.read_json(io.StringIO(counts_json), orient="split")467 468 if tab == "tab-volcano":469 return dbc.Row([470 dbc.Col(dbc.Card(dbc.CardBody([471 html.H6("Filters", className="text-uppercase mb-3",472 style={"color":ACCENT,"letterSpacing":"0.08em","fontSize":"0.75rem"}),473 html.Label("log₂FC threshold", className="small text-muted"),474 dcc.Slider(id="fc-slider", min=0, max=4, step=0.1, value=1,475 marks={i:str(i) for i in range(5)}, tooltip={"placement":"bottom"}),476 html.Label("Adjusted p-value", className="small text-muted mt-3"),477 dcc.Dropdown(id="pval-drop",478 options=[{"label":f"p < {v}","value":v}479 for v in [0.001,0.01,0.05,0.1]],480 value=0.05, clearable=False,481 style={"background":"#0f172a","color":"#e2e8f0"}),482 html.Label("Gene search (highlight)", className="small text-muted mt-3"),483 dcc.Input(id="gene-search", type="text", placeholder="e.g. TP53, BRCA1",484 className="form-control form-control-sm",485 style={"background":"#0f172a","color":"#e2e8f0","border":"1px solid #334155"}),486 html.Hr(style={"borderColor":"#334155"}),487 html.H6("Summary", className="text-uppercase mb-2",488 style={"color":ACCENT,"letterSpacing":"0.08em","fontSize":"0.75rem"}),489 html.Div(id="volcano-summary"),490 ]), style={"background":CARD}), md=2),491 dbc.Col([492 dcc.Graph(id="volcano-plot", style={"height":"460px"},493 config={"displayModeBar":True,"modeBarButtonsToRemove":["lasso2d"]}),494 dbc.Row([495 dbc.Col(html.H6("Selected / Significant Genes",496 className="mt-1 mb-0 text-muted small"), width="auto"),497 dbc.Col(dbc.Button(498 [html.I(className="bi bi-download me-1"), "Export CSV"],499 id="export-btn", size="sm", color="secondary", outline=True,500 ), width="auto"),501 ], className="mt-3 mb-2 align-items-center"),502 dcc.Download(id="export-download"),503 html.Div(id="de-table-container"),504 ], md=10),505 ], className="g-3")506 507 elif tab == "tab-ma":508 n_up = int((df["regulation"]=="Up").sum())509 n_dn = int((df["regulation"]=="Down").sum())510 return dbc.Row([dbc.Col([511 dcc.Graph(figure=build_ma(df), style={"height":"480px"},512 config={"displayModeBar":True}),513 dbc.Row([514 dbc.Col(html.Div([html.Span("Upregulated: ", className="text-muted small"),515 html.Span(str(n_up), style={"color":"#e63946","fontWeight":"700"})]), md=3),516 dbc.Col(html.Div([html.Span("Downregulated: ", className="text-muted small"),517 html.Span(str(n_dn), style={"color":"#457b9d","fontWeight":"700"})]), md=3),518 ], className="mt-2 px-2"),519 dbc.Alert([html.I(className="bi bi-info-circle me-2"),520 "MA plot shows mean expression (X) vs fold change (Y). "521 "Genes with high expression and large fold change (top/bottom right) "522 "are the most reliable hits. Low-expression genes with high fold change "523 "(top/bottom left) may be noise."],524 color="secondary", className="mt-3 small py-2"),525 ])])526 527 elif tab == "tab-pca":528 fig, var1, var2 = build_pca(counts)529 n_ctrl = sum("Ctrl" in c for c in counts.columns)530 n_treated = sum("Treat" in c for c in counts.columns)531 return dbc.Row([dbc.Col([532 dcc.Graph(figure=fig, style={"height":"480px"}, config={"displayModeBar":True}),533 dbc.Row([534 dbc.Col(html.Div([html.Span("PC1 variance: ", className="text-muted small"),535 html.Span(f"{var1:.1f}%", style={"color":ACCENT,"fontWeight":"700"})]), md=3),536 dbc.Col(html.Div([html.Span("PC2 variance: ", className="text-muted small"),537 html.Span(f"{var2:.1f}%", style={"color":ACCENT,"fontWeight":"700"})]), md=3),538 dbc.Col(html.Div([html.Span("Control: ", className="text-muted small"),539 html.Span(str(n_ctrl), style={"color":"#457b9d","fontWeight":"700"})]), md=3),540 dbc.Col(html.Div([html.Span("Treated: ", className="text-muted small"),541 html.Span(str(n_treated), style={"color":"#e63946","fontWeight":"700"})]), md=3),542 ], className="mt-2 px-2"),543 dbc.Alert([html.I(className="bi bi-info-circle me-2"),544 "Good QC: Control samples cluster together (blue), Treated samples cluster "545 "together (red), with clear separation between groups. Outlier samples "546 "that don't cluster with their group should be investigated. "547 "PCA here is simulated from DE results — upload a raw count matrix for real PCA."],548 color="secondary", className="mt-3 small py-2"),549 ])])550 551 elif tab == "tab-correlation":552 return dbc.Row([dbc.Col([553 dcc.Graph(figure=build_sample_corr(counts), style={"height":"540px"},554 config={"displayModeBar":True}),555 dbc.Alert([html.I(className="bi bi-info-circle me-2"),556 "Values close to 1.0 = highly similar samples. "557 "Replicates within the same group (e.g. Ctrl_1 vs Ctrl_2) should have "558 "r > 0.95. Low correlation within a group may indicate a failed sample. "559 "Cross-group correlation (Control vs Treated) is expected to be lower."],560 color="secondary", className="mt-2 small py-2"),561 ])])562 563 elif tab == "tab-heatmap":564 n_top = min(40, max(int(df["significant"].sum()), 20))565 return dbc.Row([dbc.Col([566 dcc.Graph(figure=build_heatmap(df, counts, n_top=n_top),567 style={"height":"650px"}, config={"displayModeBar":True}),568 dbc.Alert([html.I(className="bi bi-info-circle me-2"),569 "Top significant genes by adjusted p-value. "570 "Red = high expression, Blue = low (Z-scored per gene). "571 "Genes are clustered by expression similarity — "572 "genes in the same cluster behave similarly across samples."],573 color="secondary", className="mt-2 small py-2"),574 ])])575 576 elif tab == "tab-pathway":577 results = _pathway_enrichment(df)578 sig_count = int(df["significant"].sum())579 n_enriched = int((results["max_score"] > 0).sum())580 return dbc.Row([dbc.Col([581 dcc.Graph(figure=build_pathway_bar(results), style={"height":"650px"},582 config={"displayModeBar":True}),583 dbc.Row([584 dbc.Col(html.Div([html.Span("Significant genes tested: ", className="text-muted small"),585 html.Span(str(sig_count), style={"color":ACCENT,"fontWeight":"700"})]), md=4),586 dbc.Col(html.Div([html.Span("Pathways with hits: ", className="text-muted small"),587 html.Span(str(n_enriched), style={"color":"#2ecc71","fontWeight":"700"})]), md=4),588 ], className="mt-2 px-2"),589 html.H6("Enrichment Summary", className="mt-3 mb-1 text-muted small"),590 _pathway_table(results),591 dbc.Alert([html.I(className="bi bi-info-circle me-2"),592 "Enrichment score = fraction of pathway genes that are significant in your data "593 "divided by the background rate. Score > 1 means the pathway is enriched. "594 "Uses 22 representative KEGG/Reactome pathways. "595 "For full pathway analysis, run clusterProfiler in R after DESeq2."],596 color="secondary", className="mt-2 small py-2"),597 ])])598 599 600# ─────────────────────────────────────────────601# CALLBACKS602# ─────────────────────────────────────────────603 604@app.callback(Output("stat-cards","children"), Input("store-de","data"))605def update_stats(de_json):606 df = pd.read_json(io.StringIO(de_json), orient="split")607 return [608 stat_card("arrow-up-circle", int((df["regulation"]=="Up").sum()), "Upregulated", "#e63946"),609 stat_card("arrow-down-circle", int((df["regulation"]=="Down").sum()), "Downregulated", "#457b9d"),610 stat_card("check2-circle", int(df["significant"].sum()), "Significant (p<0.05)","#2ecc71"),611 stat_card("bar-chart", f"{len(df):,}", "Total Genes", ACCENT),612 ]613 614 615@app.callback(616 Output("store-de","data"),617 Output("store-counts","data"),618 Output("upload-status","children"),619 Input("upload-data","contents"),620 State("upload-data","filename"),621 prevent_initial_call=True,622)623def handle_upload(contents, filename):624 if not contents:625 return dash.no_update, dash.no_update, ""626 _, content_string = contents.split(",")627 decoded = base64.b64decode(content_string)628 try:629 sep = "\t" if (filename or "").endswith(".tsv") else ","630 df = pd.read_csv(io.StringIO(decoded.decode("utf-8")), sep=sep)631 rename = {}632 for col in df.columns:633 lc = col.lower()634 if lc in ("gene","gene_id","geneid","symbol"): rename[col]="gene"635 elif lc in ("log2foldchange","log2fc","lfc"): rename[col]="log2FoldChange"636 elif lc in ("pvalue","pval","p.value","p_value"): rename[col]="pvalue"637 elif lc in ("padj","p.adj","fdr","adjusted_pvalue"): rename[col]="padj"638 elif lc in ("basemean","mean_expr","avgexpr"): rename[col]="baseMean"639 df = df.rename(columns=rename)640 missing = {"gene","log2FoldChange","pvalue","padj"} - set(df.columns)641 if missing:642 return dash.no_update, dash.no_update, dbc.Alert(643 f"Missing columns: {missing}", color="danger", className="py-1")644 df = _annotate_de(df)645 counts = make_demo_counts(df)646 return df.to_json(orient="split"), counts.to_json(orient="split"), dbc.Alert(647 f"Loaded {filename}: {len(df):,} genes", color="success", className="py-1")648 except Exception as e:649 return dash.no_update, dash.no_update, dbc.Alert(650 f"Error: {e}", color="danger", className="py-1")651 652 653@app.callback(654 Output("volcano-plot","figure"),655 Output("volcano-summary","children"),656 Input("fc-slider","value"),657 Input("pval-drop","value"),658 Input("gene-search","value"),659 Input("store-de","data"),660)661def update_volcano(fc, pval, search, de_json):662 df = pd.read_json(io.StringIO(de_json), orient="split")663 highlights = [g.strip() for g in (search or "").split(",") if g.strip()]664 fig = build_volcano(df, fc_thresh=fc, pval_thresh=pval, highlight_genes=highlights)665 n_up = int(((df["padj"]<pval)&(df["log2FoldChange"]> fc)).sum())666 n_dn = int(((df["padj"]<pval)&(df["log2FoldChange"]<-fc)).sum())667 summary = [668 html.Div([html.Span("▲ Up: ", style={"color":"#e63946"}), html.Span(str(n_up))], className="small mb-1"),669 html.Div([html.Span("▼ Down: ", style={"color":"#457b9d"}), html.Span(str(n_dn))], className="small"),670 ]671 return fig, summary672 673 674@app.callback(675 Output("de-table-container","children"),676 Input("volcano-plot","selectedData"),677 Input("fc-slider","value"),678 Input("pval-drop","value"),679 Input("store-de","data"),680)681def update_de_table(selected, fc, pval, de_json):682 df = pd.read_json(io.StringIO(de_json), orient="split")683 df["regulation"] = "NS"684 df.loc[(df["padj"]<pval)&(df["log2FoldChange"]> fc),"regulation"] = "Up"685 df.loc[(df["padj"]<pval)&(df["log2FoldChange"]<-fc),"regulation"] = "Down"686 if selected and selected.get("points"):687 sub = df[df["gene"].isin([p["text"] for p in selected["points"] if "text" in p])]688 else:689 sub = df[df["regulation"]!="NS"].nlargest(50,"-log10padj")690 sub = sub[["gene","log2FoldChange","pvalue","padj","baseMean","regulation"]].copy()691 sub["log2FoldChange"] = sub["log2FoldChange"].round(3)692 sub["pvalue"] = sub["pvalue"].apply(lambda x: f"{x:.2e}")693 sub["padj"] = sub["padj"].apply(lambda x: f"{x:.2e}")694 sub["baseMean"] = sub["baseMean"].round(1)695 return dash_table.DataTable(696 data=sub.to_dict("records"),697 columns=[{"name":c,"id":c,"type":"text"} for c in sub.columns],698 filter_action="native", sort_action="native", page_size=12,699 style_table={"overflowX":"auto"},700 style_cell={"background":CARD,"color":"#e2e8f0","border":"1px solid #334155",701 "fontSize":"0.8rem","padding":"6px 10px","textAlign":"left"},702 style_header={"background":"#0f172a","fontWeight":"700","color":ACCENT,703 "border":"1px solid #334155"},704 style_data_conditional=[705 {"if":{"filter_query":'{regulation} = "Up"', "column_id":"regulation"},"color":"#e63946","fontWeight":"700"},706 {"if":{"filter_query":'{regulation} = "Down"', "column_id":"regulation"},"color":"#457b9d","fontWeight":"700"},707 {"if":{"row_index":"odd"},"background":"#162032"},708 ],709 style_filter={"background":"#0f172a","color":"#e2e8f0","border":"1px solid #334155"},710 )711 712 713@app.callback(714 Output("export-download","data"),715 Input("export-btn","n_clicks"),716 State("store-de","data"),717 State("fc-slider","value"),718 State("pval-drop","value"),719 prevent_initial_call=True,720)721def export_csv(_, de_json, fc, pval):722 df = pd.read_json(io.StringIO(de_json), orient="split")723 df["regulation"] = "NS"724 df.loc[(df["padj"]<pval)&(df["log2FoldChange"]> fc),"regulation"] = "Up"725 df.loc[(df["padj"]<pval)&(df["log2FoldChange"]<-fc),"regulation"] = "Down"726 sig = df[df["regulation"]!="NS"].sort_values("-log10padj", ascending=False)727 return dcc.send_data_frame(728 sig[["gene","log2FoldChange","pvalue","padj","baseMean","regulation"]].to_csv,729 "significant_genes.csv", index=False,730 )731 732 733# ─────────────────────────────────────────────734if __name__ == "__main__":735 import os736 port = int(os.environ.get("PORT", 7860))737 app.run(debug=False, host="0.0.0.0", port=port)738 