OxonTechnologies/CSS_EDA_Dashboard
0
1"""2Feature classification module for detecting data types in DataFrames.3 4This module provides the DetectFeatureClasses class which automatically5classifies features as Binary, Categorical, or Continuous based on their6statistical properties.7"""8 9import numpy as np10 11 12class DetectFeatureClasses:13 """14 A class to detect feature classes in a pandas DataFrame.15 Parameters:16 ----------17 dataframe : pd.DataFrame18 The input DataFrame containing features to be classified.19 categorical_threshold : float, optional20 The relative threshold to determine if a feature is categorical based on the ratio of unique values to total rows. Default is 0.5.21 string_data_policy : str, optional22 Policy for handling string data that cannot be converted to float. Options are 'drop' to drop such features or 'ignore' to leave them as is. Default is 'drop'.23 Methods:24 -------25 feature_classes() -> dict26 Classifies features into 'Binary', 'Categorical', or 'Continuous' and returns a dictionary with feature names as keys and their classes as values.27 """28 29 def __init__(self, dataframe, categorical_threshold=0.5, string_data_policy='drop'):30 31 """32 Initializes the DetectFeatureClasses with the provided DataFrame and parameters.33 Parameters:34 ----------35 dataframe : pd.DataFrame36 The input DataFrame containing features to be classified.37 categorical_threshold : float, optional38 The relative threshold to determine if a feature is categorical based on the ratio of unique values to total rows. Default is 0.5.39 string_data_policy : str, optional40 Policy for handling string data that cannot be converted to float. Options are 'drop' to drop such features or 'ignore' to leave them as is. Default is 'drop'.41 """42 43 self.dataframe = dataframe44 self.categorical_threshold = categorical_threshold45 self.string_data_policy = string_data_policy46 47 def _binaries(self):48 """49 Identifies binary features in the DataFrame.50 51 A feature is considered binary if it has at most 2 unique values.52 53 Returns54 -------55 list56 A list of column names that are classified as binary features.57 """58 binary_columns = [column for column in self.dataframe.columns if len(self.dataframe[column].unique()) <= 2]59 return binary_columns60 61 def _categorical(self):62 """63 Identifies categorical features in the DataFrame based on the categorical threshold.64 65 A feature is considered categorical if the number of unique values66 is significantly less than the total number of rows (using the67 categorical_threshold as a relative tolerance).68 69 Returns70 -------71 list72 A list of column names that are classified as categorical features.73 """74 categorical_columns = []75 for column in self.dataframe.columns:76 # Check if unique count is not close to total rows (within threshold)77 if np.isclose(len(78 self.dataframe[column].unique()), len(self.dataframe), 79 rtol=self.categorical_threshold80 ) is False:81 categorical_columns.append(column)82 return categorical_columns83 84 def feature_classes(self):85 """86 Classifies features in the DataFrame into 'Binary', 'Categorical', or 'Continuous'.87 Returns:88 -------89 dict90 A dictionary with feature names as keys and their classes ('Binary', 'Categorical', 'Continuous') as values.91 list92 A list of features that were dropped due to string data policy.93 """94 binary_columns = self._binaries()95 categorical_columns = self._categorical()96 features_class_types = {}97 excess_columns = []98 99 # Classify each feature100 for feature in self.dataframe.columns:101 if feature in binary_columns:102 features_class_types[feature] = 'Binary'103 elif feature in categorical_columns:104 features_class_types[feature] = 'Categorical'105 else:106 # Try to convert to float to determine if continuous107 try: 108 self.dataframe[feature] = self.dataframe[feature].astype(float)109 features_class_types[feature] = 'Continuous'110 except ValueError:111 # Cannot convert to float - handle based on policy112 if self.string_data_policy == 'drop':113 excess_columns.append(feature)114 else:115 # 'ignore' policy: leave as-is (not recommended)116 pass117 118 return features_class_types, excess_columns119 