Team Ai
Apppublic

xHunterW/Data_Patterns_Module_3_Disc

sourceHugging Facemitupdated 4mo agoView on Hugging Face
0likes
app.py115 linesDownload Raw Back to root
1import streamlit as st2import pandas as pd3import matplotlib.pyplot as plt4import os5 6st.title("Boston Housing Dataset Explorer")7 8st.write("""9The **Boston Housing Dataset** contains information collected by the U.S. Census Service10about housing in the Boston, MA area. Each row represents a suburb or town, and the target11variable `MEDV` is the **median home value in thousands of dollars**. This app lets you12explore how different neighborhood characteristics relate to housing prices.13""")14 15st.markdown("---")16 17# Load dataset18df = pd.read_csv(os.path.join(os.path.dirname(__file__), "Boston-house-price-data.csv"))19 20COLUMN_DESCRIPTIONS = {21    "CRIM":    "Per capita crime rate by town",22    "ZN":      "Proportion of residential land zoned for lots over 25,000 sq ft",23    "INDUS":   "Proportion of non-retail business acres per town",24    "CHAS":    "Charles River dummy variable (1 if tract bounds river, 0 otherwise)",25    "NOX":     "Nitric oxide concentration (parts per 10 million)",26    "RM":      "Average number of rooms per dwelling",27    "AGE":     "Proportion of owner-occupied units built prior to 1940",28    "DIS":     "Weighted distances to five Boston employment centres",29    "RAD":     "Index of accessibility to radial highways",30    "TAX":     "Full-value property tax rate per $10,000",31    "PTRATIO": "Pupil-teacher ratio by town",32    "B":       "1000(Bk - 0.63)^2, where Bk is the proportion of Black residents",33    "LSTAT":   "Percentage of lower-status population",34    "MEDV":    "Median value of owner-occupied homes in $1,000s (target variable)",35}36 37# ── Section 1: Dataset Preview ────────────────────────────────────────────────38st.header("1. Dataset Preview")39 40show_all = st.checkbox("Show full dataset (all 506 rows)", value=False)41if show_all:42    st.dataframe(df, use_container_width=True)43else:44    st.dataframe(df.head(20), use_container_width=True)45 46with st.expander("Column Descriptions"):47    for col, desc in COLUMN_DESCRIPTIONS.items():48        st.markdown(f"- **{col}**: {desc}")49 50st.markdown("---")51 52# ── Section 2: Filter by Home Value (Slider) ─────────────────────────────────53st.header("2. Filter by Median Home Value")54 55st.write("Use the slider below to filter suburbs by their median home value (`MEDV`).")56 57min_val = float(df["MEDV"].min())58max_val = float(df["MEDV"].max())59 60price_range = st.slider(61    "Select MEDV range ($1,000s)",62    min_value=min_val,63    max_value=max_val,64    value=(min_val, max_val),65    step=0.5,66)67 68filtered_df = df[(df["MEDV"] >= price_range[0]) & (df["MEDV"] <= price_range[1])]69 70st.write(f"**{len(filtered_df)}** suburbs match this price range.")71st.dataframe(filtered_df, use_container_width=True)72 73st.markdown("---")74 75# ── Section 3: Explore a Variable (Dropdown) ─────────────────────────────────76st.header("3. Explore a Variable")77 78st.write(79    "Select any column to see its distribution and how it relates to median home value."80)81 82numeric_cols = [c for c in df.columns if c != "MEDV"]83selected_col = st.selectbox("Choose a variable:", numeric_cols)84 85col_a, col_b = st.columns(2)86 87with col_a:88    st.subheader(f"Distribution of {selected_col}")89    fig1, ax1 = plt.subplots()90    ax1.hist(filtered_df[selected_col], bins=20, color="#4C72B0", edgecolor="white")91    ax1.set_xlabel(selected_col)92    ax1.set_ylabel("Count")93    st.pyplot(fig1)94    plt.close(fig1)95 96with col_b:97    st.subheader(f"{selected_col} vs. MEDV")98    fig2, ax2 = plt.subplots()99    ax2.scatter(filtered_df[selected_col], filtered_df["MEDV"], alpha=0.5, color="#DD8452")100    ax2.set_xlabel(selected_col)101    ax2.set_ylabel("MEDV ($1,000s)")102    st.pyplot(fig2)103    plt.close(fig2)104 105if selected_col in COLUMN_DESCRIPTIONS:106    st.caption(f"**{selected_col}**: {COLUMN_DESCRIPTIONS[selected_col]}")107 108st.markdown("---")109 110# ── Section 4: Summary Statistics ────────────────────────────────────────────111st.header("4. Summary Statistics")112 113st.write("Statistics computed on the currently filtered subset of data.")114st.dataframe(filtered_df.describe().T.style.format("{:.2f}"), use_container_width=True)115