OxonTechnologies/CSS_EDA_Dashboard
0
1"""2Correlation matrix generation module for mixed data types.3 4This module provides the CorrelationMatrixGenerator class which computes5correlation/association matrices for DataFrames containing mixed data types6(Continuous, Binary, Categorical). It automatically selects appropriate7correlation measures based on feature type pairs.8"""9 10import numpy as np11import pandas as pd12from scipy.stats import chi2_contingency, pointbiserialr13from tqdm import tqdm14 15 16class CorrelationMatrixGenerator:17 """18 A class to generate a correlation/association matrix for a pandas DataFrame,19 handling different data types appropriately. It supports Continuous, Binary, and Categorical data types.20 Parameters:21 ----------22 df : pd.DataFrame23 The input DataFrame containing features for correlation analysis.24 feature_classes : dict25 A dictionary mapping column names to their data types ('Continuous', 'Binary', 'Categorical').26 continuous_vs_continuous_method : str, optional27 Method to use for estimating the correlation coefficient of two continuous data types. Default is 'pearson'.28 Methods:29 -------30 generate_matrix() -> pd.DataFrame31 Generates and returns a symmetric correlation/association matrix for the DataFrame.32 """33 34 def __init__(self, df, feature_classes, continuous_vs_continuous_method='pearson'):35 36 """37 Initialize with a DataFrame and a dictionary mapping column names to data types.38 39 Parameters:40 df : pandas.DataFrame41 The DataFrame containing your data.42 feature_classes : dict43 A dictionary where keys are column names in df and values are their data types.44 Valid types are 'Continuous', 'Binary', or 'Categorical'.45 continuous_vs_continuous_method : str46 Method to use for estimating the correlation coefficient of two continuous data47 """48 49 self.df = df50 self.feature_classes = feature_classes51 self.continuous_vs_continuous_method = continuous_vs_continuous_method52 53 54 @staticmethod55 def recode_binary(series):56 """57 Ensure a binary series is coded as 0 and 1.58 59 If the series is already numeric with values {0,1}, it is returned as is.60 Otherwise, it maps the two unique values to 0 and 1.61 62 Parameters63 ----------64 series : pd.Series65 A binary series to recode.66 67 Returns68 -------69 pd.Series70 Binary series with values {0, 1}.71 72 Raises73 ------74 ValueError75 If the series does not appear to be binary (has more than 2 unique values).76 """77 # Check if already numeric and in {0, 1}78 if pd.api.types.is_numeric_dtype(series):79 unique_vals = series.dropna().unique()80 if set(unique_vals) <= {0, 1}:81 return series82 # Map two unique values to {0, 1}83 unique_vals = series.dropna().unique()84 if len(unique_vals) == 2:85 mapping = {unique_vals[0]: 0, unique_vals[1]: 1}86 return series.map(mapping)87 else:88 raise ValueError("Series does not appear to be binary")89 90 @staticmethod91 def cramers_v(x, y):92 """93 Calculate Cramér's V statistic for a categorical-categorical association.94 95 Cramér's V is a measure of association between two nominal variables,96 ranging from 0 (no association) to 1 (perfect association).97 98 Parameters99 ----------100 x, y : array-like101 Two categorical variables.102 103 Returns104 -------105 float106 Cramér's V statistic, or np.nan if computation is not possible.107 """108 contingency_table = pd.crosstab(x, y)109 chi2 = chi2_contingency(contingency_table)[0]110 n = contingency_table.values.sum()111 min_dim = min(contingency_table.shape) - 1112 if n == 0 or min_dim == 0:113 return np.nan114 return np.sqrt(chi2 / (n * min_dim))115 116 @staticmethod117 def anova_eta(categories, measurements):118 """119 Compute the eta (η) as an effect size measure derived from one-way ANOVA.120 It indicates the proportion of variance in the continuous variable (measurements)121 explained by the categorical grouping (categories). Higher values indicate a stronger effect.122 123 Parameters:124 categories : array-like (categorical grouping)125 measurements : array-like (continuous values)126 127 Returns:128 eta : float, between 0 and 1 representing the effect size.129 """130 131 # Factorize the categorical variable132 factors, _ = pd.factorize(categories)133 categories_count = np.max(factors) + 1134 overall_mean = np.mean(measurements)135 ss_between = 0.0 # Sum of Squares136 137 for i in range(categories_count):138 group = measurements[factors == i]139 n_i = len(group)140 if n_i == 0:141 continue142 group_mean = np.mean(group)143 ss_between += n_i * ((group_mean - overall_mean) ** 2)144 145 ss_total = np.sum((measurements - overall_mean) ** 2)146 147 if ss_total == 0:148 return np.nan149 150 eta = np.sqrt(ss_between / ss_total)151 152 return eta153 154 def compute_pairwise_correlation(self, series1, type1, series2, type2):155 """156 Compute the correlation/association between two series based on their data types.157 158 Parameters:159 series1, series2 : pandas.Series160 type1, type2 : str, one of 'Continuous', 'Binary', 'Categorical'161 162 Returns:163 A correlation/association measure (float) or np.nan if not defined.164 """165 166 # ------------- Homogeneous Data types -------------167 168 # Continuous vs. Continuous: Pearson correlation169 if {type1, type2} == {'Continuous', 'Continuous'}:170 return series1.corr(series2, method=self.continuous_vs_continuous_method)171 172 # Binary vs. Binary: Phi coefficient (using Pearson on recoded binaries)173 elif {type1, type2} == {'Binary', 'Binary'}:174 try:175 s1 = self.recode_binary(series1)176 s2 = self.recode_binary(series2)177 except Exception as e:178 return np.nan179 return s1.corr(s2, method='pearson')180 181 # Categorical vs. Categorical: Use Cramér's V182 elif {type1, type2} == {'Categorical', 'Categorical'}:183 return self.cramers_v(series1, series2)184 185 # ------------- Heterogeneous Data Types -------------186 187 # Binary & Continuous: Point-biserial correlation coefficient188 elif {type1, type2} == {'Continuous', 'Binary'}:189 190 binary_series = series1 if type1 == 'Binary' else series2191 continuous_series = series2 if type2 == 'Continuous' else series1192 193 try:194 binary_series = self.recode_binary(binary_series)195 except Exception as e:196 return np.nan197 198 corr, _ = pointbiserialr(binary_series, continuous_series)199 200 return corr201 202 # Categorical vs. Continuous: Use ANOVA-based effect size (η)203 elif {type1, type2} == {'Continuous', 'Categorical'}:204 return self.anova_eta(series1, series2) if type1 == 'Categorical' else self.anova_eta(series2, series1)205 206 # Binary vs. Categorical: Treat as nominal and use Cramér's V207 elif {type1, type2} == {'Binary', 'Categorical'}:208 return self.cramers_v(series1, series2)209 210 else:211 return np.nan212 213 def generate_matrix(self):214 """215 Generate a symmetric correlation/association matrix for the specified columns,216 using the appropriate method based on their data types.217 218 The matrix is computed by iterating over all feature pairs and selecting219 the appropriate correlation measure based on their types. The matrix220 is symmetric (corr(A, B) = corr(B, A)).221 222 Returns223 -------224 pd.DataFrame225 A symmetric correlation/association matrix with feature names as226 both index and columns. Values are rounded to 4 decimal places.227 """228 factors = list(self.feature_classes.keys())229 corr_matrix = pd.DataFrame(index=factors, columns=factors, dtype=float)230 231 # Compute pairwise correlations232 for i, var1 in tqdm(list(enumerate(factors))):233 for j, var2 in enumerate(factors):234 if i == j:235 # Diagonal: perfect correlation with itself236 corr_matrix.loc[var1, var2] = 1.0237 elif pd.isna(corr_matrix.loc[var1, var2]):238 # Compute correlation only if not already computed (upper triangle)239 series1 = self.df[var1]240 series2 = self.df[var2]241 type1 = self.feature_classes[var1]242 type2 = self.feature_classes[var2]243 corr_value = self.compute_pairwise_correlation(series1, type1, series2, type2)244 # Fill both upper and lower triangle for symmetry245 corr_matrix.loc[var1, var2] = corr_value246 corr_matrix.loc[var2, var1] = corr_value # ensure symmetry247 248 return corr_matrix.round(4)249 