pranee31/emailClassification
0
1import pandas as pd2import os3from sklearn.model_selection import train_test_split4from sklearn.ensemble import RandomForestClassifier5from sklearn.pipeline import Pipeline6from sklearn.feature_extraction.text import TfidfVectorizer7from sklearn.metrics import classification_report8import joblib9 10def train_classifier(csv_path: str, model_path: str = "saved_model/email_classifier.pkl"):11 df = pd.read_csv(csv_path)12 if 'email' not in df.columns or 'type' not in df.columns:13 raise ValueError("CSV must have 'email' and 'type' columns.")14 15 X = df['email']16 y = df['type']17 18 X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)19 20 clf_pipeline = Pipeline([21 ('tfidf', TfidfVectorizer(stop_words='english')),22 ('clf', RandomForestClassifier(n_estimators=100, random_state=42))23 ])24 25 clf_pipeline.fit(X_train, y_train)26 27 preds = clf_pipeline.predict(X_test)28 print("\nClassification Report:\n", classification_report(y_test, preds))29 30 os.makedirs(os.path.dirname(model_path), exist_ok=True)31 joblib.dump(clf_pipeline, model_path)32 print(f"Model saved to {model_path}")