Team Ai
Apppublic

OxonTechnologies/CSS_EDA_Dashboard

sourceHugging Faceupdated 5mo agoView on Hugging Face
0likes
feature_class.py119 linesDownload Raw Back to utils
1"""2Feature classification module for detecting data types in DataFrames.3 4This module provides the DetectFeatureClasses class which automatically5classifies features as Binary, Categorical, or Continuous based on their6statistical properties.7"""8 9import numpy as np10 11 12class DetectFeatureClasses:13    """14    A class to detect feature classes in a pandas DataFrame.15    Parameters:16    ----------17    dataframe : pd.DataFrame18        The input DataFrame containing features to be classified.19    categorical_threshold : float, optional20        The relative threshold to determine if a feature is categorical based on the ratio of unique values to total rows. Default is 0.5.21    string_data_policy : str, optional22        Policy for handling string data that cannot be converted to float. Options are 'drop' to drop such features or 'ignore' to leave them as is. Default is 'drop'.23    Methods:24    -------25    feature_classes() -> dict26        Classifies features into 'Binary', 'Categorical', or 'Continuous' and returns a dictionary with feature names as keys and their classes as values.27    """28 29    def __init__(self, dataframe, categorical_threshold=0.5, string_data_policy='drop'):30 31        """32        Initializes the DetectFeatureClasses with the provided DataFrame and parameters.33        Parameters:34        ----------35        dataframe : pd.DataFrame36            The input DataFrame containing features to be classified.37        categorical_threshold : float, optional38            The relative threshold to determine if a feature is categorical based on the ratio of unique values to total rows. Default is 0.5.39        string_data_policy : str, optional40            Policy for handling string data that cannot be converted to float. Options are 'drop' to drop such features or 'ignore' to leave them as is. Default is 'drop'.41        """42 43        self.dataframe = dataframe44        self.categorical_threshold = categorical_threshold45        self.string_data_policy = string_data_policy46    47    def _binaries(self):48        """49        Identifies binary features in the DataFrame.50        51        A feature is considered binary if it has at most 2 unique values.52        53        Returns54        -------55        list56            A list of column names that are classified as binary features.57        """58        binary_columns = [column for column in self.dataframe.columns if len(self.dataframe[column].unique()) <= 2]59        return binary_columns60    61    def _categorical(self):62        """63        Identifies categorical features in the DataFrame based on the categorical threshold.64        65        A feature is considered categorical if the number of unique values66        is significantly less than the total number of rows (using the67        categorical_threshold as a relative tolerance).68        69        Returns70        -------71        list72            A list of column names that are classified as categorical features.73        """74        categorical_columns = []75        for column in self.dataframe.columns:76            # Check if unique count is not close to total rows (within threshold)77            if np.isclose(len(78                self.dataframe[column].unique()), len(self.dataframe), 79                rtol=self.categorical_threshold80                ) is False:81                categorical_columns.append(column)82        return categorical_columns83 84    def feature_classes(self):85        """86        Classifies features in the DataFrame into 'Binary', 'Categorical', or 'Continuous'.87        Returns:88        -------89        dict90            A dictionary with feature names as keys and their classes ('Binary', 'Categorical', 'Continuous') as values.91        list92            A list of features that were dropped due to string data policy.93        """94        binary_columns = self._binaries()95        categorical_columns = self._categorical()96        features_class_types = {}97        excess_columns = []98 99        # Classify each feature100        for feature in self.dataframe.columns:101            if feature in binary_columns:102                features_class_types[feature] = 'Binary'103            elif feature in categorical_columns:104                features_class_types[feature] = 'Categorical'105            else:106                # Try to convert to float to determine if continuous107                try: 108                    self.dataframe[feature] = self.dataframe[feature].astype(float)109                    features_class_types[feature] = 'Continuous'110                except ValueError:111                    # Cannot convert to float - handle based on policy112                    if self.string_data_policy == 'drop':113                        excess_columns.append(feature)114                    else:115                        # 'ignore' policy: leave as-is (not recommended)116                        pass117 118        return features_class_types, excess_columns119