Team Ai
Apppublic

PriyaPbs/ModelOptix

sourceHugging Faceupdated 6mo agoView on Hugging Face
0likes
preprocess.py94 linesDownload Raw Back to ml_pipeline
1import numpy as np2import pandas as pd3from sklearn.compose import ColumnTransformer4from sklearn.preprocessing import FunctionTransformer5from sklearn.pipeline import Pipeline6from sklearn.preprocessing import StandardScaler, OneHotEncoder, LabelEncoder7from sklearn.impute import SimpleImputer8 9 10def _looks_like_datetime(series: pd.Series) -> bool:11    if not pd.api.types.is_object_dtype(series) and not pd.api.types.is_string_dtype(series):12        return False13    parsed = pd.to_datetime(series, errors="coerce")14    return parsed.notna().mean() >= 0.815 16 17def _expand_datetime(values):18    series = pd.Series(values.squeeze() if getattr(values, "ndim", 1) > 1 else values)19    parsed = pd.to_datetime(series, errors="coerce")20    frame = pd.DataFrame({21        "year": parsed.dt.year.fillna(0),22        "month": parsed.dt.month.fillna(0),23        "day": parsed.dt.day.fillna(0),24        "dayofweek": parsed.dt.dayofweek.fillna(0),25        "hour": parsed.dt.hour.fillna(0),26    })27    return frame.astype(float).to_numpy()28 29 30def build_preprocessor(df: pd.DataFrame, target: str):31    X = df.drop(columns=[target])32 33    row_count = max(len(X), 1)34    drop_cols = []35    for col in X.columns:36        unique_ratio = X[col].nunique(dropna=False) / row_count37        lower = str(col).lower()38        if X[col].nunique(dropna=False) <= 1:39            drop_cols.append(col)40        elif unique_ratio > 0.95 and (lower.endswith("_id") or lower == "id" or "attempt_id" in lower):41            drop_cols.append(col)42        elif lower.endswith("_name") or lower == "name":43            drop_cols.append(col)44 45    if drop_cols:46        X = X.drop(columns=drop_cols)47 48    date_cols = [col for col in X.columns if _looks_like_datetime(X[col])]49    num_cols = X.select_dtypes(include=["number"]).columns.tolist()50    cat_cols = [col for col in X.select_dtypes(include=["object", "category", "bool"]).columns.tolist() if col not in date_cols]51 52    print(f"[Preprocessor] Numeric cols: {num_cols}")53    print(f"[Preprocessor] Datetime cols: {date_cols}")54    print(f"[Preprocessor] Categorical cols: {cat_cols}")55    print(f"[Preprocessor] Dropped cols: {drop_cols}")56 57    if not num_cols and not cat_cols and not date_cols:58        raise ValueError("No usable feature columns found.")59 60    transformers = []61 62    if num_cols:63        num_pipeline = Pipeline([64            ("imputer", SimpleImputer(strategy="median")),65            ("scaler", StandardScaler()),66        ])67        transformers.append(("num", num_pipeline, num_cols))68 69    if cat_cols:70        cat_pipeline = Pipeline([71            ("imputer", SimpleImputer(strategy="most_frequent")),72            ("encoder", OneHotEncoder(handle_unknown="ignore", sparse_output=False)),73        ])74        transformers.append(("cat", cat_pipeline, cat_cols))75 76    if date_cols:77        date_pipeline = Pipeline([78            ("imputer", SimpleImputer(strategy="most_frequent")),79            ("expand", FunctionTransformer(_expand_datetime, validate=False)),80            ("scaler", StandardScaler()),81        ])82        transformers.append(("date", date_pipeline, date_cols))83 84    return ColumnTransformer(transformers=transformers, remainder="drop")85 86 87def encode_target(y: pd.Series):88    if pd.api.types.is_numeric_dtype(y):89        return y, None90    le = LabelEncoder()91    y_encoded = pd.Series(le.fit_transform(y.astype(str)), index=y.index, name=y.name)92    print(f"[Preprocessor] Target encoded: {dict(zip(le.classes_, le.transform(le.classes_)))}")93    return y_encoded, le94