Team Ai
Apppublic

TangibleAI/mathtext-fastapi

sourceHugging Faceagpl-3.0updated 3y agoView on Hugging Face
1likes
intent_classification.py56 linesDownload Raw Back to mathtext_fastapi
1import numpy as np2import pandas as pd3 4from pathlib import Path5from sentence_transformers import SentenceTransformer6from sklearn.linear_model import LogisticRegression7from joblib import dump, load8 9def pickle_model(model):10    DATA_DIR = Path(__file__).parent.parent / "mathtext_fastapi" / "data" / "intent_classification_model.joblib"11    dump(model, DATA_DIR)12 13 14def create_intent_classification_model():15    encoder = SentenceTransformer('all-MiniLM-L6-v2')16    # path = list(Path.cwd().glob('*.csv'))17    DATA_DIR = Path(__file__).parent.parent / "mathtext_fastapi" / "data" / "labeled_data.csv"18 19    print("DATA_DIR")20    print(f"{DATA_DIR}")21 22    with open(f"{DATA_DIR}",'r', newline='', encoding='utf-8') as f:23        df = pd.read_csv(f)24    df = df[df.columns[:2]]25    df = df.dropna()26    X_explore = np.array([list(encoder.encode(x)) for x in df['Utterance']])27    X = np.array([list(encoder.encode(x)) for x in df['Utterance']])28    y = df['Label']29    model = LogisticRegression(class_weight='balanced')30    model.fit(X, y, sample_weight=None)31 32    print("MODEL")33    print(model)34 35    pickle_model(model)36 37 38def retrieve_intent_classification_model():39    DATA_DIR = Path(__file__).parent.parent / "mathtext_fastapi" / "data" / "intent_classification_model.joblib"40    model = load(DATA_DIR)41    return model42 43 44encoder = SentenceTransformer('all-MiniLM-L6-v2')45# model = retrieve_intent_classification_model()46DATA_DIR = Path(__file__).parent.parent / "mathtext_fastapi" / "data" / "intent_classification_model.joblib"47model = load(DATA_DIR)48 49 50def predict_message_intent(message):51    tokenized_utterance = np.array([list(encoder.encode(message))])52    predicted_label = model.predict(tokenized_utterance)53    predicted_probabilities = model.predict_proba(tokenized_utterance)54    confidence_score = predicted_probabilities.max()55 56    return {"type": "intent", "data": predicted_label[0], "confidence": confidence_score}