xHunterW/Data_Patterns_Module_3_Disc
0
1import streamlit as st2import pandas as pd3import matplotlib.pyplot as plt4import os5 6st.title("Boston Housing Dataset Explorer")7 8st.write("""9The **Boston Housing Dataset** contains information collected by the U.S. Census Service10about housing in the Boston, MA area. Each row represents a suburb or town, and the target11variable `MEDV` is the **median home value in thousands of dollars**. This app lets you12explore how different neighborhood characteristics relate to housing prices.13""")14 15st.markdown("---")16 17# Load dataset18df = pd.read_csv(os.path.join(os.path.dirname(__file__), "Boston-house-price-data.csv"))19 20COLUMN_DESCRIPTIONS = {21 "CRIM": "Per capita crime rate by town",22 "ZN": "Proportion of residential land zoned for lots over 25,000 sq ft",23 "INDUS": "Proportion of non-retail business acres per town",24 "CHAS": "Charles River dummy variable (1 if tract bounds river, 0 otherwise)",25 "NOX": "Nitric oxide concentration (parts per 10 million)",26 "RM": "Average number of rooms per dwelling",27 "AGE": "Proportion of owner-occupied units built prior to 1940",28 "DIS": "Weighted distances to five Boston employment centres",29 "RAD": "Index of accessibility to radial highways",30 "TAX": "Full-value property tax rate per $10,000",31 "PTRATIO": "Pupil-teacher ratio by town",32 "B": "1000(Bk - 0.63)^2, where Bk is the proportion of Black residents",33 "LSTAT": "Percentage of lower-status population",34 "MEDV": "Median value of owner-occupied homes in $1,000s (target variable)",35}36 37# ── Section 1: Dataset Preview ────────────────────────────────────────────────38st.header("1. Dataset Preview")39 40show_all = st.checkbox("Show full dataset (all 506 rows)", value=False)41if show_all:42 st.dataframe(df, use_container_width=True)43else:44 st.dataframe(df.head(20), use_container_width=True)45 46with st.expander("Column Descriptions"):47 for col, desc in COLUMN_DESCRIPTIONS.items():48 st.markdown(f"- **{col}**: {desc}")49 50st.markdown("---")51 52# ── Section 2: Filter by Home Value (Slider) ─────────────────────────────────53st.header("2. Filter by Median Home Value")54 55st.write("Use the slider below to filter suburbs by their median home value (`MEDV`).")56 57min_val = float(df["MEDV"].min())58max_val = float(df["MEDV"].max())59 60price_range = st.slider(61 "Select MEDV range ($1,000s)",62 min_value=min_val,63 max_value=max_val,64 value=(min_val, max_val),65 step=0.5,66)67 68filtered_df = df[(df["MEDV"] >= price_range[0]) & (df["MEDV"] <= price_range[1])]69 70st.write(f"**{len(filtered_df)}** suburbs match this price range.")71st.dataframe(filtered_df, use_container_width=True)72 73st.markdown("---")74 75# ── Section 3: Explore a Variable (Dropdown) ─────────────────────────────────76st.header("3. Explore a Variable")77 78st.write(79 "Select any column to see its distribution and how it relates to median home value."80)81 82numeric_cols = [c for c in df.columns if c != "MEDV"]83selected_col = st.selectbox("Choose a variable:", numeric_cols)84 85col_a, col_b = st.columns(2)86 87with col_a:88 st.subheader(f"Distribution of {selected_col}")89 fig1, ax1 = plt.subplots()90 ax1.hist(filtered_df[selected_col], bins=20, color="#4C72B0", edgecolor="white")91 ax1.set_xlabel(selected_col)92 ax1.set_ylabel("Count")93 st.pyplot(fig1)94 plt.close(fig1)95 96with col_b:97 st.subheader(f"{selected_col} vs. MEDV")98 fig2, ax2 = plt.subplots()99 ax2.scatter(filtered_df[selected_col], filtered_df["MEDV"], alpha=0.5, color="#DD8452")100 ax2.set_xlabel(selected_col)101 ax2.set_ylabel("MEDV ($1,000s)")102 st.pyplot(fig2)103 plt.close(fig2)104 105if selected_col in COLUMN_DESCRIPTIONS:106 st.caption(f"**{selected_col}**: {COLUMN_DESCRIPTIONS[selected_col]}")107 108st.markdown("---")109 110# ── Section 4: Summary Statistics ────────────────────────────────────────────111st.header("4. Summary Statistics")112 113st.write("Statistics computed on the currently filtered subset of data.")114st.dataframe(filtered_df.describe().T.style.format("{:.2f}"), use_container_width=True)115 