isabel/testing-streamlit
1
1### ----------------------------- ###2### libraries ###3### ----------------------------- ###4 5import streamlit as st6import pickle as pkl7import pandas as pd8import numpy as np9from sklearn.model_selection import train_test_split10from sklearn.linear_model import LogisticRegression11from sklearn import metrics12 13 14### ----------------------------- ###15### interface setup ###16### ----------------------------- ###17 18with open('styles.css') as f:19 st.markdown(f'<style>{f.read()}</style>', unsafe_allow_html=True)20 21 22### ------------------------------ ###23### data transformation ###24### ------------------------------ ###25 26# load dataset27uncleaned_data = pd.read_csv('data.csv')28 29# remove timestamp from dataset (always first column)30uncleaned_data = uncleaned_data.iloc[: , 1:]31data = pd.DataFrame()32 33# keep track of which columns are categorical and what 34# those columns' value mappings are35# structure: {colname1: {...}, colname2: {...} }36cat_value_dicts = {}37final_colname = uncleaned_data.columns[len(uncleaned_data.columns) - 1]38 39# for each column...40for (colname, colval) in uncleaned_data.iteritems():41 42 # check if col is already a number; if so, add col directly43 # to new dataframe and skip to next column44 if isinstance(colval.values[0], (np.integer, float)):45 data[colname] = uncleaned_data[colname].copy()46 continue47 48 # structure: {0: "lilac", 1: "blue", ...}49 new_dict = {}50 val = 0 # first index per column51 transformed_col_vals = [] # new numeric datapoints52 53 # if not, for each item in that column...54 for (row, item) in enumerate(colval.values):55 56 # if item is not in this col's dict...57 if item not in new_dict:58 new_dict[item] = val59 val += 160 61 # then add numerical value to transformed dataframe62 transformed_col_vals.append(new_dict[item])63 64 # reverse dictionary only for final col (0, 1) => (vals)65 if colname == final_colname:66 new_dict = {value : key for (key, value) in new_dict.items()}67 68 cat_value_dicts[colname] = new_dict69 data[colname] = transformed_col_vals70 71 72### -------------------------------- ###73### model training ###74### -------------------------------- ###75 76def train_model():77 # select features and prediction; automatically selects last column as prediction78 cols = len(data.columns)79 num_features = cols - 180 x = data.iloc[: , :num_features]81 y = data.iloc[: , num_features:]82 83 # split data into training and testing sets84 x_train, x_test, y_train, y_test = train_test_split(x, y, test_size=0.25)85 86 # instantiate the model (using default parameters)87 model = LogisticRegression()88 model.fit(x_train, y_train.values.ravel())89 y_pred = model.predict(x_test)90 91 # save the model to file using the pickle package92 with open('model.pkl', 'wb') as f:93 pkl.dump(model, f)94 95 # save model accuracy to file using the pickle package96 with open('acc.txt', 'w+') as f:97 acc = metrics.accuracy_score(y_test, y_pred)98 f.write(str(round(acc * 100, 1)) + '%')99 100 return model101 102### -------------------------------- ###103### rerun logic ###104### -------------------------------- ###105 106# check to see if this is the first time running the script,107# if the model has already been trained and saved, load it108try:109 with open('model.pkl', 'rb') as f:110 model = pkl.load(f)111 112# if this is the first time running the script, train the model113# and save it to the file model.pkl114except FileNotFoundError as e:115 model = train_model()116 117# read the model accuracy from file118with open('acc.txt', 'r') as f:119 acc = f.read()120 121 122### ------------------------------- ###123### interface creation ###124### ------------------------------- ###125 126# uses the logistic regression to predict for a generic number127# of features128def general_predictor(input_list):129 features = []130 131 # transform categorical input132 for colname, input in zip(data.columns, input_list):133 if (colname in cat_value_dicts):134 features.append(cat_value_dicts[colname][input])135 else:136 features.append(input)137 138 # predict single datapoint139 new_input = [features]140 result = model.predict(new_input)141 return cat_value_dicts[final_colname][result[0]]142 143def get_feat():144 feats = [abs(x) for x in model.coef_[0]]145 max_val = max(feats)146 idx = feats.index(max_val)147 return data.columns[idx]148 149with open('info.md') as f:150 st.title(f.readline())151 st.subheader('Take the quiz to get a personalized recommendation using AI.')152 153form = st.form('ml-inputs')154 155# add data labels to replace those lost via star-args156inputls = []157for colname in data.columns:158 # skip last column159 if colname == final_colname:160 continue161 162 # access categories dict if data is categorical163 # otherwise, just use a number input164 if colname in cat_value_dicts:165 radio_options = list(cat_value_dicts[colname].keys())166 inputls.append(form.selectbox(colname, radio_options))167 else:168 # add numerical input169 inputls.append(form.number_imput(colname))170 171# generate gradio interface172if form.form_submit_button("Submit to get your recommendation!"):173 prediction = general_predictor(inputls)174 175 form.subheader(prediction)176 177col1, col2 = st.columns(2)178col1.metric("Number of Different Possible Results", len(cat_value_dicts[final_colname]))179col2.metric("Model Accuracy", acc)180st.metric("Most Important Question", "")181st.subheader(get_feat())182st.markdown("***")183 184with open('info.md') as f:185 f.readline()186 st.markdown(f.read())