MasanneckLab/Withings_Normalization_App
1
1import streamlit as st2import normalizer_model3import numpy as np4import pandas as pd5import altair as alt6import plotly.graph_objects as go7from scipy.stats import norm8 9# Configure the Streamlit page before other commands10st.set_page_config(11 page_title="Smartwatch Normative Z-Score Calculator",12 layout="wide",13)14 15 16# Cache the normative DataFrame load17def load_norm_df(path: str):18 return normalizer_model.load_normative_table(path)19 20 21load_norm_df = st.cache_data(load_norm_df)22 23# Load dataset24norm_df = load_norm_df("Table_1_summary_measure.csv")25 26# Friendly biomarker labels27BIOMARKER_LABELS = {28 "nb_steps": "Number of Steps",29 "max_steps": "Maximum Steps",30 "mean_active_time": "Mean Active Time",31 "sbp": "Systolic Blood Pressure",32 "dbp": "Diastolic Blood Pressure",33 "sleep_duration": "Sleep Duration",34 "avg_night_hr": "Average Night Heart Rate",35 "nb_moderate_active_minutes": "Moderate Active Minutes",36 "nb_vigorous_active_minutes": "Vigorous Active Minutes",37 "weight": "Weight",38 "pwv": "Pulse Wave Velocity",39 # add any others here40}41 42# Biomarkers temporarily disabled in the UI. Remove from this set to re-enable.43DISABLED_BIOMARKERS = {"weight", "sbp", "dbp", "pwv", "nb_vigorous_active_minutes"}44 45 46def main():47 if "disclaimer_shown" not in st.session_state:48 st.info(49 "These calculations are dedicated for scientific purposes only. "50 "For detailed questions regarding personal health data contact your "51 "healthcare professionals."52 )53 st.session_state.disclaimer_shown = True54 st.title("Smartwatch Normative Z-Score Calculator")55 st.sidebar.header("Input Parameters")56 57 # Region with default Western Europe58 regions = sorted(norm_df["area"].unique())59 if "Western Europe" in regions:60 default_region = "Western Europe"61 else:62 default_region = regions[0]63 region = st.sidebar.selectbox(64 "Region",65 regions,66 index=regions.index(default_region),67 )68 69 # Gender selection70 gender = st.sidebar.selectbox(71 "Gender",72 sorted(norm_df["gender"].unique()),73 )74 75 # Age input: choose between years or group76 st.sidebar.subheader("Age Input")77 age_input_mode = st.sidebar.radio(78 "Age input mode",79 ("Years", "Group"),80 )81 if age_input_mode == "Years":82 age_years = st.sidebar.number_input(83 "Age (years)",84 min_value=0,85 max_value=120,86 value=30,87 step=1,88 )89 age_param = age_years90 else:91 age_groups = sorted(92 norm_df["Age"].unique(),93 key=lambda x: int(x.split("-")[0]),94 )95 age_group = st.sidebar.selectbox("Age group", [""] + age_groups)96 age_param = age_group97 98 # BMI input: choose between value or category99 st.sidebar.subheader("BMI Input")100 bmi_input_mode = st.sidebar.radio(101 "BMI input mode",102 ("Value", "Category"),103 )104 if bmi_input_mode == "Value":105 bmi_val = st.sidebar.number_input(106 "BMI",107 min_value=0.0,108 max_value=100.0,109 value=24.0,110 step=0.1,111 format="%.1f",112 )113 bmi_param = bmi_val114 else:115 bmi_cats = sorted(norm_df["Bmi"].unique())116 bmi_cat = st.sidebar.selectbox("BMI category", [""] + bmi_cats)117 bmi_param = bmi_cat118 119 # Biomarker selection with friendly labels120 codes = sorted(121 c for c in norm_df["Biomarkers"].unique() if c not in DISABLED_BIOMARKERS122 )123 friendly = [BIOMARKER_LABELS.get(c, c.title()) for c in codes]124 default_idx = friendly.index("Number of Steps")125 selected_label = st.sidebar.selectbox(126 "Biomarker",127 friendly,128 index=default_idx,129 )130 biomarker = codes[friendly.index(selected_label)]131 132 # Value input with consistent float types133 default_value = 6500.0 if biomarker == "nb_steps" else 0.0134 # Determine upper bound from normative data135 mask = norm_df["Biomarkers"].str.lower() == biomarker.lower()136 max_val = float(norm_df.loc[mask, "max"].max())137 value = st.sidebar.number_input(138 f"{selected_label} value",139 min_value=0.0,140 max_value=max_val,141 value=default_value,142 step=1.0,143 )144 145 # Compute146 norm_button = st.sidebar.button("Compute Normative Z-Score")147 if norm_button:148 try:149 res = normalizer_model.compute_normative_position(150 value=value,151 biomarker=biomarker,152 age_group=age_param,153 region=region,154 gender=gender,155 bmi=bmi_param,156 normative_df=norm_df,157 )158 except Exception as e:159 st.error(f"Error: {e}")160 return161 162 # Show metrics163 st.subheader("Results")164 m1, m2, m3, m4, m5 = st.columns(5)165 m1.metric("Z-Score", f"{res['z_score']:.2f}")166 m2.metric("Percentile", f"{res['percentile']:.2f}")167 m3.metric("Mean", f"{res['mean']:.2f}")168 m4.metric("SD", f"{res['sd']:.2f}")169 m5.metric("Sample Size", res["n"])170 171 # Compute actual age group and BMI category for cohort summary172 age_group_str = normalizer_model._categorize_age(age_param, norm_df)173 bmi_cat = normalizer_model.categorize_bmi(bmi_param)174 st.markdown(175 f"**Basis of calculation:** Data from region **{region}**, "176 f"gender **{gender}**, age group **{age_group_str}**, "177 f"and BMI category **{bmi_cat}. "178 f"Sample size: {res['n']}**."179 )180 181 # Detailed statistics table182 st.subheader("Detailed Statistics")183 stats_df = pd.DataFrame(184 {185 "Statistic": [186 "Z-Score",187 "Percentile",188 "Mean",189 "SD",190 "Sample Size",191 "Median",192 "Q1",193 "Q3",194 "IQR",195 "MAD",196 "SE",197 "CI",198 ],199 "Value": [200 f"{res['z_score']:.2f}",201 f"{res['percentile']:.2f}",202 f"{res['mean']:.2f}",203 f"{res['sd']:.2f}",204 res.get("n", "N/A"),205 f"{res.get('median', float('nan')):.2f}",206 f"{res.get('q1', float('nan')):.2f}",207 f"{res.get('q3', float('nan')):.2f}",208 f"{res.get('iqr', float('nan')):.2f}",209 f"{res.get('mad', float('nan')):.2f}",210 f"{res.get('se', float('nan')):.2f}",211 f"{res.get('ci', float('nan')):.2f}",212 ],213 }214 )215 st.table(stats_df)216 217 # Normality assumption note218 note = (219 "*Note: Percentile and z-score estimation assume a normal "220 "distribution based on global Withings user data stratified by "221 "the parameters entered.*"222 )223 st.write(note)224 225 # Normality checks226 import normality_checks as nc227 228 R = nc.iqr_tail_heaviness(res["iqr"], res["sd"])229 q1_z, q3_z = nc.quartile_z_scores(230 res["mean"],231 res["sd"],232 res["q1"],233 res["q3"],234 )235 skew = nc.pearson_skewness(res["mean"], res["median"], res["sd"])236 st.subheader("Normality Heuristics")237 238 # Determine skewness interpretation239 if abs(skew) <= 0.1:240 skew_interp = "Symmetric (OK)"241 elif abs(skew) <= 0.5:242 skew_interp = f"{'Right' if skew > 0 else 'Left'} slight skew (usually OK)"243 elif abs(skew) <= 1.0:244 skew_interp = f"{'Right' if skew > 0 else 'Left'} noticeable skew"245 else:246 skew_interp = f"{'Right' if skew > 0 else 'Left'} strong skew"247 248 norm_checks = pd.DataFrame(249 {250 "Check": [251 "IQR/SD",252 "Q1 z-score",253 "Q3 z-score",254 "Pearson Skewness",255 ],256 "Value": [257 f"{R:.2f}",258 f"{q1_z:.2f}",259 f"{q3_z:.2f}",260 f"{skew:.2f}",261 ],262 "Flag": [263 (264 "Heavier tails"265 if R > 1.5266 else "Lighter tails" if R < 1.2 else "OK"267 ),268 "Deviation" if abs(q1_z + 0.6745) > 0.1 else "OK",269 "Deviation" if abs(q3_z - 0.6745) > 0.1 else "OK",270 skew_interp,271 ],272 }273 )274 st.table(norm_checks)275 276 # Add skewness interpretation guide277 st.markdown(278 """279 **Pearson Skewness Interpretation:**280 - ≈ 0: Symmetric distribution281 - ±0.1 to ±0.5: Slight/moderate skew282 - ±0.5 to ±1: Noticeable skew283 - larger than±1: Strong skew284 285 - Positive values: Right skew (longer tail on right)286 - Negative values: Left skew (longer tail on left)287 """288 )289 290 # Warning if heuristic checks indicate non-normality291 if any(("OK" not in val) for val in norm_checks["Flag"]):292 st.warning(293 "Warning: Heuristic checks indicate possible deviations "294 "from normality; interpret z-score and percentiles with "295 "caution."296 )297 298 # Skew-Corrected Results (optional)299 with st.expander("Optional: Skew-Corrected Results"):300 st.write("Adjusts for skew via Pearson Type III back-transform.")301 st.write("Error often <1 percentile point when |skew| ≤ 0.5.")302 st.write("Usually more useful for stronger skewed distributions.")303 st.write("Note: This is a heuristic and may not always be accurate.")304 res_skew = normalizer_model.compute_skew_corrected_position(305 value=value,306 mean=res["mean"],307 sd=res["sd"],308 median=res["median"],309 )310 pct_skew = f"{res_skew['percentile_skew_corrected']:.2f}"311 sc1, sc2 = st.columns(2)312 sc1.metric(313 "Skew-Corrected Z-Score",314 f"{res_skew['z_skew_corrected']:.2f}",315 )316 sc2.metric(317 "Skew-Corrected Percentile",318 pct_skew,319 )320 321 st.markdown("---")322 st.subheader("Visualizations")323 # Prepare data for normal distribution324 z_vals = np.linspace(-4, 4, 400)325 density = norm.pdf(z_vals)326 df_chart = pd.DataFrame({"z": z_vals, "density": density})327 # Shade area up to observed z-score328 area = (329 alt.Chart(df_chart)330 .mark_area(color="orange", opacity=0.3)331 .transform_filter(alt.datum.z <= res["z_score"])332 .encode(333 x=alt.X(334 "z:Q",335 title="z-score",336 ),337 y=alt.Y(338 "density:Q",339 title="Density",340 ),341 )342 )343 # Plot distribution line344 line = (345 alt.Chart(df_chart)346 .mark_line(color="orange")347 .encode(348 x="z:Q",349 y="density:Q",350 )351 )352 # Vertical line at observed z353 vline = (354 alt.Chart(pd.DataFrame({"z": [res["z_score"]]}))355 .mark_rule(color="orange")356 .encode(x="z:Q")357 )358 chart = (area + line + vline).properties(359 width=600,360 height=300,361 title="Standard Normal Distribution",362 )363 st.altair_chart(chart, use_container_width=True)364 # Text summary365 st.write(366 f"Your value is z = {res['z_score']:.2f}, which places you in "367 f"the {res['percentile']:.1f}th percentile of a normal "368 f"distribution."369 )370 # Bullet chart showing z-score location371 # Using a horizontal bullet gauge from -3 to 3 SD372 bullet = go.Figure(373 go.Indicator(374 mode="number+gauge",375 value=res["z_score"],376 number={"suffix": " SD"},377 gauge={378 "shape": "bullet",379 "axis": {380 "range": [-3, 3],381 "tickmode": "linear",382 "dtick": 0.5,383 },384 "bar": {"color": "orange"},385 },386 )387 )388 bullet.update_layout(389 height=150,390 margin={"t": 20, "b": 20, "l": 20, "r": 20},391 )392 st.plotly_chart(bullet, use_container_width=True)393 # Show percentile text394 st.write(f"Percentile: {res['percentile']:.1f}%")395 else:396 st.sidebar.info(397 "Fill in all inputs and click Compute " "to get normative Z-score."398 )399 400 # Z-Score Classification Guide (always visible)401 st.markdown("---")402 with st.expander("📊 Z-Score Classification Guide"):403 st.markdown("""404 **How to interpret Z-Scores:**405 406 | Z-Score Range | Classification | Percentile Range |407 |:-------------:|:--------------:|:----------------:|408 | z < -2.0 | Very Low | < 2.3% |409 | -2.0 ≤ z < -0.5 | Below Average | 2.3% - 30.9% |410 | **-0.5 ≤ z < 0.5** | **Average** | **30.9% - 69.1%** |411 | 0.5 ≤ z < 2.0 | Above Average | 69.1% - 97.7% |412 | z ≥ 2.0 | Very High | > 97.7% |413 414 **Context matters:**415 - For **steps, sleep duration, and active minutes**: Higher values are generally better ✓416 - For **heart rate**: Lower resting values are generally better ✓417 418 *A z-score of 0 means you are exactly at the population average for your demographic group.*419 """)420 421 # Footer422 st.markdown("---")423 st.markdown(424 "Built with ❤️ in Düsseldorf. © Lars Masanneck 2026. "425 "Thanks to Withings for sharing this data openly."426 )427 st.markdown(428 "*This tool is part of the publication "429 "\"Population-Normalised Wearable Metrics Quantify Real-World Disability "430 "in Multiple Sclerosis\" currently in review.*"431 )432 433 434if __name__ == "__main__":435 main()436 