Team Ai
Apppublic

MasanneckLab/Withings_Normalization_App

sourceHugging Faceupdated 9mo agoView on Hugging Face
1likes
Z-Score_Calculator.py436 linesDownload Raw Back to root
1import streamlit as st2import normalizer_model3import numpy as np4import pandas as pd5import altair as alt6import plotly.graph_objects as go7from scipy.stats import norm8 9# Configure the Streamlit page before other commands10st.set_page_config(11    page_title="Smartwatch Normative Z-Score Calculator",12    layout="wide",13)14 15 16# Cache the normative DataFrame load17def load_norm_df(path: str):18    return normalizer_model.load_normative_table(path)19 20 21load_norm_df = st.cache_data(load_norm_df)22 23# Load dataset24norm_df = load_norm_df("Table_1_summary_measure.csv")25 26# Friendly biomarker labels27BIOMARKER_LABELS = {28    "nb_steps": "Number of Steps",29    "max_steps": "Maximum Steps",30    "mean_active_time": "Mean Active Time",31    "sbp": "Systolic Blood Pressure",32    "dbp": "Diastolic Blood Pressure",33    "sleep_duration": "Sleep Duration",34    "avg_night_hr": "Average Night Heart Rate",35    "nb_moderate_active_minutes": "Moderate Active Minutes",36    "nb_vigorous_active_minutes": "Vigorous Active Minutes",37    "weight": "Weight",38    "pwv": "Pulse Wave Velocity",39    # add any others here40}41 42# Biomarkers temporarily disabled in the UI. Remove from this set to re-enable.43DISABLED_BIOMARKERS = {"weight", "sbp", "dbp", "pwv", "nb_vigorous_active_minutes"}44 45 46def main():47    if "disclaimer_shown" not in st.session_state:48        st.info(49            "These calculations are dedicated for scientific purposes only. "50            "For detailed questions regarding personal health data contact your "51            "healthcare professionals."52        )53        st.session_state.disclaimer_shown = True54    st.title("Smartwatch Normative Z-Score Calculator")55    st.sidebar.header("Input Parameters")56 57    # Region with default Western Europe58    regions = sorted(norm_df["area"].unique())59    if "Western Europe" in regions:60        default_region = "Western Europe"61    else:62        default_region = regions[0]63    region = st.sidebar.selectbox(64        "Region",65        regions,66        index=regions.index(default_region),67    )68 69    # Gender selection70    gender = st.sidebar.selectbox(71        "Gender",72        sorted(norm_df["gender"].unique()),73    )74 75    # Age input: choose between years or group76    st.sidebar.subheader("Age Input")77    age_input_mode = st.sidebar.radio(78        "Age input mode",79        ("Years", "Group"),80    )81    if age_input_mode == "Years":82        age_years = st.sidebar.number_input(83            "Age (years)",84            min_value=0,85            max_value=120,86            value=30,87            step=1,88        )89        age_param = age_years90    else:91        age_groups = sorted(92            norm_df["Age"].unique(),93            key=lambda x: int(x.split("-")[0]),94        )95        age_group = st.sidebar.selectbox("Age group", [""] + age_groups)96        age_param = age_group97 98    # BMI input: choose between value or category99    st.sidebar.subheader("BMI Input")100    bmi_input_mode = st.sidebar.radio(101        "BMI input mode",102        ("Value", "Category"),103    )104    if bmi_input_mode == "Value":105        bmi_val = st.sidebar.number_input(106            "BMI",107            min_value=0.0,108            max_value=100.0,109            value=24.0,110            step=0.1,111            format="%.1f",112        )113        bmi_param = bmi_val114    else:115        bmi_cats = sorted(norm_df["Bmi"].unique())116        bmi_cat = st.sidebar.selectbox("BMI category", [""] + bmi_cats)117        bmi_param = bmi_cat118 119    # Biomarker selection with friendly labels120    codes = sorted(121        c for c in norm_df["Biomarkers"].unique() if c not in DISABLED_BIOMARKERS122    )123    friendly = [BIOMARKER_LABELS.get(c, c.title()) for c in codes]124    default_idx = friendly.index("Number of Steps")125    selected_label = st.sidebar.selectbox(126        "Biomarker",127        friendly,128        index=default_idx,129    )130    biomarker = codes[friendly.index(selected_label)]131 132    # Value input with consistent float types133    default_value = 6500.0 if biomarker == "nb_steps" else 0.0134    # Determine upper bound from normative data135    mask = norm_df["Biomarkers"].str.lower() == biomarker.lower()136    max_val = float(norm_df.loc[mask, "max"].max())137    value = st.sidebar.number_input(138        f"{selected_label} value",139        min_value=0.0,140        max_value=max_val,141        value=default_value,142        step=1.0,143    )144 145    # Compute146    norm_button = st.sidebar.button("Compute Normative Z-Score")147    if norm_button:148        try:149            res = normalizer_model.compute_normative_position(150                value=value,151                biomarker=biomarker,152                age_group=age_param,153                region=region,154                gender=gender,155                bmi=bmi_param,156                normative_df=norm_df,157            )158        except Exception as e:159            st.error(f"Error: {e}")160            return161 162        # Show metrics163        st.subheader("Results")164        m1, m2, m3, m4, m5 = st.columns(5)165        m1.metric("Z-Score", f"{res['z_score']:.2f}")166        m2.metric("Percentile", f"{res['percentile']:.2f}")167        m3.metric("Mean", f"{res['mean']:.2f}")168        m4.metric("SD", f"{res['sd']:.2f}")169        m5.metric("Sample Size", res["n"])170 171        # Compute actual age group and BMI category for cohort summary172        age_group_str = normalizer_model._categorize_age(age_param, norm_df)173        bmi_cat = normalizer_model.categorize_bmi(bmi_param)174        st.markdown(175            f"**Basis of calculation:** Data from region **{region}**, "176            f"gender **{gender}**, age group **{age_group_str}**, "177            f"and BMI category **{bmi_cat}. "178            f"Sample size: {res['n']}**."179        )180 181        # Detailed statistics table182        st.subheader("Detailed Statistics")183        stats_df = pd.DataFrame(184            {185                "Statistic": [186                    "Z-Score",187                    "Percentile",188                    "Mean",189                    "SD",190                    "Sample Size",191                    "Median",192                    "Q1",193                    "Q3",194                    "IQR",195                    "MAD",196                    "SE",197                    "CI",198                ],199                "Value": [200                    f"{res['z_score']:.2f}",201                    f"{res['percentile']:.2f}",202                    f"{res['mean']:.2f}",203                    f"{res['sd']:.2f}",204                    res.get("n", "N/A"),205                    f"{res.get('median', float('nan')):.2f}",206                    f"{res.get('q1', float('nan')):.2f}",207                    f"{res.get('q3', float('nan')):.2f}",208                    f"{res.get('iqr', float('nan')):.2f}",209                    f"{res.get('mad', float('nan')):.2f}",210                    f"{res.get('se', float('nan')):.2f}",211                    f"{res.get('ci', float('nan')):.2f}",212                ],213            }214        )215        st.table(stats_df)216 217        # Normality assumption note218        note = (219            "*Note: Percentile and z-score estimation assume a normal "220            "distribution based on global Withings user data stratified by "221            "the parameters entered.*"222        )223        st.write(note)224 225        # Normality checks226        import normality_checks as nc227 228        R = nc.iqr_tail_heaviness(res["iqr"], res["sd"])229        q1_z, q3_z = nc.quartile_z_scores(230            res["mean"],231            res["sd"],232            res["q1"],233            res["q3"],234        )235        skew = nc.pearson_skewness(res["mean"], res["median"], res["sd"])236        st.subheader("Normality Heuristics")237 238        # Determine skewness interpretation239        if abs(skew) <= 0.1:240            skew_interp = "Symmetric (OK)"241        elif abs(skew) <= 0.5:242            skew_interp = f"{'Right' if skew > 0 else 'Left'} slight skew (usually OK)"243        elif abs(skew) <= 1.0:244            skew_interp = f"{'Right' if skew > 0 else 'Left'} noticeable skew"245        else:246            skew_interp = f"{'Right' if skew > 0 else 'Left'} strong skew"247 248        norm_checks = pd.DataFrame(249            {250                "Check": [251                    "IQR/SD",252                    "Q1 z-score",253                    "Q3 z-score",254                    "Pearson Skewness",255                ],256                "Value": [257                    f"{R:.2f}",258                    f"{q1_z:.2f}",259                    f"{q3_z:.2f}",260                    f"{skew:.2f}",261                ],262                "Flag": [263                    (264                        "Heavier tails"265                        if R > 1.5266                        else "Lighter tails" if R < 1.2 else "OK"267                    ),268                    "Deviation" if abs(q1_z + 0.6745) > 0.1 else "OK",269                    "Deviation" if abs(q3_z - 0.6745) > 0.1 else "OK",270                    skew_interp,271                ],272            }273        )274        st.table(norm_checks)275 276        # Add skewness interpretation guide277        st.markdown(278            """279        **Pearson Skewness Interpretation:**280        - ≈ 0: Symmetric distribution281        - ±0.1 to ±0.5: Slight/moderate skew282        - ±0.5 to ±1: Noticeable skew283        - larger than±1: Strong skew284 285        - Positive values: Right skew (longer tail on right)286        - Negative values: Left skew (longer tail on left)287        """288        )289 290        # Warning if heuristic checks indicate non-normality291        if any(("OK" not in val) for val in norm_checks["Flag"]):292            st.warning(293                "Warning: Heuristic checks indicate possible deviations "294                "from normality; interpret z-score and percentiles with "295                "caution."296            )297 298        # Skew-Corrected Results (optional)299        with st.expander("Optional: Skew-Corrected Results"):300            st.write("Adjusts for skew via Pearson Type III back-transform.")301            st.write("Error often <1 percentile point when |skew| ≤ 0.5.")302            st.write("Usually more useful for stronger skewed distributions.")303            st.write("Note: This is a heuristic and may not always be accurate.")304            res_skew = normalizer_model.compute_skew_corrected_position(305                value=value,306                mean=res["mean"],307                sd=res["sd"],308                median=res["median"],309            )310            pct_skew = f"{res_skew['percentile_skew_corrected']:.2f}"311            sc1, sc2 = st.columns(2)312            sc1.metric(313                "Skew-Corrected Z-Score",314                f"{res_skew['z_skew_corrected']:.2f}",315            )316            sc2.metric(317                "Skew-Corrected Percentile",318                pct_skew,319            )320 321        st.markdown("---")322        st.subheader("Visualizations")323        # Prepare data for normal distribution324        z_vals = np.linspace(-4, 4, 400)325        density = norm.pdf(z_vals)326        df_chart = pd.DataFrame({"z": z_vals, "density": density})327        # Shade area up to observed z-score328        area = (329            alt.Chart(df_chart)330            .mark_area(color="orange", opacity=0.3)331            .transform_filter(alt.datum.z <= res["z_score"])332            .encode(333                x=alt.X(334                    "z:Q",335                    title="z-score",336                ),337                y=alt.Y(338                    "density:Q",339                    title="Density",340                ),341            )342        )343        # Plot distribution line344        line = (345            alt.Chart(df_chart)346            .mark_line(color="orange")347            .encode(348                x="z:Q",349                y="density:Q",350            )351        )352        # Vertical line at observed z353        vline = (354            alt.Chart(pd.DataFrame({"z": [res["z_score"]]}))355            .mark_rule(color="orange")356            .encode(x="z:Q")357        )358        chart = (area + line + vline).properties(359            width=600,360            height=300,361            title="Standard Normal Distribution",362        )363        st.altair_chart(chart, use_container_width=True)364        # Text summary365        st.write(366            f"Your value is z = {res['z_score']:.2f}, which places you in "367            f"the {res['percentile']:.1f}th percentile of a normal "368            f"distribution."369        )370        # Bullet chart showing z-score location371        # Using a horizontal bullet gauge from -3 to 3 SD372        bullet = go.Figure(373            go.Indicator(374                mode="number+gauge",375                value=res["z_score"],376                number={"suffix": " SD"},377                gauge={378                    "shape": "bullet",379                    "axis": {380                        "range": [-3, 3],381                        "tickmode": "linear",382                        "dtick": 0.5,383                    },384                    "bar": {"color": "orange"},385                },386            )387        )388        bullet.update_layout(389            height=150,390            margin={"t": 20, "b": 20, "l": 20, "r": 20},391        )392        st.plotly_chart(bullet, use_container_width=True)393        # Show percentile text394        st.write(f"Percentile: {res['percentile']:.1f}%")395    else:396        st.sidebar.info(397            "Fill in all inputs and click Compute " "to get normative Z-score."398        )399 400    # Z-Score Classification Guide (always visible)401    st.markdown("---")402    with st.expander("📊 Z-Score Classification Guide"):403        st.markdown("""404        **How to interpret Z-Scores:**405        406        | Z-Score Range | Classification | Percentile Range |407        |:-------------:|:--------------:|:----------------:|408        | z < -2.0 | Very Low | < 2.3% |409        | -2.0 ≤ z < -0.5 | Below Average | 2.3% - 30.9% |410        | **-0.5 ≤ z < 0.5** | **Average** | **30.9% - 69.1%** |411        | 0.5 ≤ z < 2.0 | Above Average | 69.1% - 97.7% |412        | z ≥ 2.0 | Very High | > 97.7% |413        414        **Context matters:**415        - For **steps, sleep duration, and active minutes**: Higher values are generally better ✓416        - For **heart rate**: Lower resting values are generally better ✓417        418        *A z-score of 0 means you are exactly at the population average for your demographic group.*419        """)420 421    # Footer422    st.markdown("---")423    st.markdown(424        "Built with ❤️ in Düsseldorf. © Lars Masanneck 2026. "425        "Thanks to Withings for sharing this data openly."426    )427    st.markdown(428        "*This tool is part of the publication "429        "\"Population-Normalised Wearable Metrics Quantify Real-World Disability "430        "in Multiple Sclerosis\" currently in review.*"431    )432 433 434if __name__ == "__main__":435    main()436