PriyaPbs/ModelOptix
0
1import numpy as np2import pandas as pd3from sklearn.compose import ColumnTransformer4from sklearn.preprocessing import FunctionTransformer5from sklearn.pipeline import Pipeline6from sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder7from sklearn.impute import SimpleImputer8 9 10def _looks_like_datetime(series: pd.Series) -> bool:11 if not pd.api.types.is_object_dtype(series) and not pd.api.types.is_string_dtype(series):12 return False13 parsed = pd.to_datetime(series, errors="coerce")14 return parsed.notna().mean() >= 0.815 16 17def _expand_datetime(values):18 series = pd.Series(values.squeeze() if getattr(values, "ndim", 1) > 1 else values)19 parsed = pd.to_datetime(series, errors="coerce")20 frame = pd.DataFrame({21 "year": parsed.dt.year.fillna(0),22 "month": parsed.dt.month.fillna(0),23 "day": parsed.dt.day.fillna(0),24 "dayofweek": parsed.dt.dayofweek.fillna(0),25 "hour": parsed.dt.hour.fillna(0),26 })27 return frame.astype(float).to_numpy()28 29 30def build_preprocessor(df: pd.DataFrame, target: str):31 X = df.drop(columns=[target])32 33 row_count = max(len(X), 1)34 drop_cols = []35 for col in X.columns:36 unique_ratio = X[col].nunique(dropna=False) / row_count37 lower = str(col).lower()38 if X[col].nunique(dropna=False) <= 1:39 drop_cols.append(col)40 elif unique_ratio > 0.95 and (lower.endswith("_id") or lower == "id" or "attempt_id" in lower):41 drop_cols.append(col)42 elif lower.endswith("_name") or lower == "name":43 drop_cols.append(col)44 45 if drop_cols:46 X = X.drop(columns=drop_cols)47 48 date_cols = [col for col in X.columns if _looks_like_datetime(X[col])]49 num_cols = X.select_dtypes(include=["number"]).columns.tolist()50 cat_cols = [col for col in X.select_dtypes(include=["object", "category", "bool"]).columns.tolist() if col not in date_cols]51 52 print(f"[Preprocessor] Numeric cols: {num_cols}")53 print(f"[Preprocessor] Datetime cols: {date_cols}")54 print(f"[Preprocessor] Categorical cols: {cat_cols}")55 print(f"[Preprocessor] Dropped cols: {drop_cols}")56 57 if not num_cols and not cat_cols and not date_cols:58 raise ValueError("No usable feature columns found.")59 60 transformers = []61 62 if num_cols:63 num_pipeline = Pipeline([64 ("imputer", SimpleImputer(strategy="median")),65 ("scaler", StandardScaler()),66 ])67 transformers.append(("num", num_pipeline, num_cols))68 69 if cat_cols:70 cat_pipeline = Pipeline([71 ("imputer", SimpleImputer(strategy="most_frequent")),72 ("encoder", OneHotEncoder(handle_unknown="ignore", sparse_output=False)),73 ])74 transformers.append(("cat", cat_pipeline, cat_cols))75 76 if date_cols:77 date_pipeline = Pipeline([78 ("imputer", SimpleImputer(strategy="most_frequent")),79 ("expand", FunctionTransformer(_expand_datetime, validate=False)),80 ("scaler", StandardScaler()),81 ])82 transformers.append(("date", date_pipeline, date_cols))83 84 return ColumnTransformer(transformers=transformers, remainder="drop")85 86 87def encode_target(y: pd.Series):88 if pd.api.types.is_numeric_dtype(y):89 return y, None90 le = LabelEncoder()91 y_encoded = pd.Series(le.fit_transform(y.astype(str)), index=y.index, name=y.name)92 print(f"[Preprocessor] Target encoded: {dict(zip(le.classes_, le.transform(le.classes_)))}")93 return y_encoded, le94 