Sj8287/Sentiment_Classification
0
1from flask import render_template,redirect,url_for,flash,request2from wtforms.validators import ValidationError3from app import app4from tensorflow.keras.preprocessing.sequence import pad_sequences5from keras.layers import Input, Dense, LSTM, GRU, Embedding6from keras.layers import Activation, Bidirectional, GlobalMaxPool1D, GlobalMaxPool2D, Dropout7from keras.models import Model8from keras.preprocessing import text, sequence9import transformers10from transformers import AutoTokenizer11from tokenizers import BertWordPieceTokenizer12from keras.initializers import Constant13import numpy as np14import re15import tensorflow as tf16import os17@app.route('/')18def home_page():19 return render_template('index.html')20 21 22tokenizer = transformers.AutoTokenizer.from_pretrained("distilbert-base-uncased")23fast_tokenizer = BertWordPieceTokenizer('distilbert_base_uncased/vocab.txt', lowercase=True)24 25 26 27def fast_encode_sentence(text, tokenizer, maxlen=128): 28 tokenizer.enable_truncation(max_length=maxlen)29 tokenizer.enable_padding(length=maxlen)30 all_ids = []31 32 text_chunk = text33 encs = tokenizer.encode(text_chunk)34 all_ids.extend([encs.ids])35 36 return np.array(all_ids)37 38 39 40 41 42transformer_layer = transformers.TFDistilBertModel.from_pretrained('distilbert-base-uncased')43 44embedding_size = 12845inp = Input(shape=(128, ))46embedding_matrix=transformer_layer.weights[0].numpy()47x = Embedding(embedding_matrix.shape[0], embedding_matrix.shape[1],embeddings_initializer=Constant(embedding_matrix),trainable=False)(inp)48x = Bidirectional(LSTM(25, return_sequences=True,recurrent_regularizer='L1L2'))(x)49x = GlobalMaxPool1D()(x)50x = Dropout(0.9)(x)51x = Dense(50, activation='relu',kernel_initializer='he_normal',kernel_regularizer="L1L2")(x)52x = Dropout(0.9)(x)53x = Dense(1, activation='sigmoid')(x)54 55model = Model(inputs=[inp], outputs=x)56model.load_weights('distilbert_model_weights.best.hdf5')57 58 59def predict_on_sentence(model,text):60 text=text.lower()61 pattern = re.compile('http[s]?://(?:[a-zA-Z]|[0-9]|[$-_@.&+]|[!*\(\),]|(?:%[0-9a-fA-F][0-9a-fA-F]))+')62 text = pattern.sub('', text)63 text = re.sub(r"i'm", "i am", text)64 text = re.sub(r"he's", "he is", text)65 text = re.sub(r"she's", "she is", text)66 text = re.sub(r"that's", "that is", text) 67 text = re.sub(r"what's", "what is", text)68 text = re.sub(r"where's", "where is", text) 69 text = re.sub(r"\'ll", " will", text) 70 text = re.sub(r"\'ve", " have", text) 71 text = re.sub(r"\'re", " are", text)72 text = re.sub(r"\'d", " would", text)73 text = re.sub(r"\'ve", " have", text)74 text = re.sub(r"won't", "will not", text)75 text = re.sub(r"don't", "do not", text)76 text = re.sub(r"did't", "did not", text)77 text = re.sub(r"can't", "can not", text)78 text = re.sub(r"it's", "it is", text)79 text = re.sub(r"couldn't", "could not", text)80 text = re.sub(r"have't", "have not", text)81 text=re.sub(r"(@[A-Za-z0-9]+)|([^0-9A-Za-z \t])|(\w+:\/\/\S+)|^rt|http.+?", "", text)82 text = re.sub(r"[,.\"!@#$%^&*(){}?/;`~:<>+=-]", "", text)83 text = re.sub(r'(.)\1{3,}',r'\1', text)84 final_text=fast_encode_sentence(text,fast_tokenizer)85 prediction=model.predict(final_text)86 final_text=tf.squeeze(tf.round(prediction))87 return final_text88 89 90@app.route('/predict',methods=['POST'])91def predict():92 int_features = request.form.get("sentence")93 int_features=str(int_features)94 final_result=predict_on_sentence(model,int_features)95 result='bad'96 if(final_result==1):97 result='good'98 return render_template('index.html', prediction_text='This is a {} comment'.format(result))99 100 