Casio991ms/MathBot
6
1# -*- coding: utf-8 -*-2"""MWP_Solver_-_Transformer_with_Multi-head_Attention_Block (1).ipynb3 4Automatically generated by Colaboratory.5 6Original file is located at7 https://colab.research.google.com/drive/1Tn_j0k8EJ7ny_h7Pjm0stJhNMG4si_y_8"""9 10# ! pip install -q gradio11 12import pandas as pd13import re14import os15import time16import random17import numpy as np18 19os.system("pip install tensorflow")20os.system("pip install scikit-learn")21os.system("pip install spacy")22os.system("pip install nltk")23os.system("spacy download en_core_web_sm")24 25import tensorflow as tf26import matplotlib.pyplot as plt27import matplotlib.ticker as ticker28from sklearn.model_selection import train_test_split29 30import pickle31 32import spacy33 34from nltk.translate.bleu_score import corpus_bleu35 36import gradio as gr37 38os.system("wget -nc 'https://docs.google.com/uc?export=download&id=1Y8Ee4lUs30BAfFtL3d3VjwChmbDG7O6H' -O data_final.pkl")39os.system('''wget --load-cookies /tmp/cookies.txt "https://docs.google.com/uc?export=download&confirm=$(wget --quiet --save-cookies /tmp/cookies.txt --keep-session-cookies --no-check-certificate 'https://docs.google.com/uc?export=download&id=1gAQVaxg_2mNcr8qwx0J2UwpkvoJgLu6a' -O- | sed -rn 's/.*confirm=([0-9A-Za-z_]+).*/\\1\\n/p')&id=1gAQVaxg_2mNcr8qwx0J2UwpkvoJgLu6a" -O checkpoints.zip && rm -rf /tmp/cookies.txt''')40os.system("unzip -n './checkpoints.zip' -d './'")41 42nlp = spacy.load("en_core_web_sm")43 44tf.__version__45 46with open('data_final.pkl', 'rb') as f:47 df = pickle.load(f)48 49df.shape50 51df.head()52 53input_exps = list(df['Question'].values)54 55def convert_eqn(eqn):56 '''57 Add a space between every character in the equation string.58 Eg: 'x = 23 + 88' becomes 'x = 2 3 + 8 8'59 '''60 elements = list(eqn)61 return ' '.join(elements)62 63target_exps = list(df['Equation'].apply(lambda x: convert_eqn(x)).values)64 65# Input: Word problem66input_exps[:5]67 68# Target: Equation69target_exps[:5]70 71len(pd.Series(input_exps)), len(pd.Series(input_exps).unique())72 73len(pd.Series(target_exps)), len(pd.Series(target_exps).unique())74 75def preprocess_input(sentence):76 '''77 For the word problem, convert everything to lowercase, add spaces around all78 punctuations and digits, and remove any extra spaces. 79 '''80 sentence = sentence.lower().strip()81 sentence = re.sub(r"([?.!,’])", r" \1 ", sentence)82 sentence = re.sub(r"([0-9])", r" \1 ", sentence)83 sentence = re.sub(r'[" "]+', " ", sentence)84 sentence = sentence.rstrip().strip()85 return sentence86 87def preprocess_target(sentence):88 '''89 For the equation, convert it to lowercase and remove extra spaces90 '''91 sentence = sentence.lower().strip()92 return sentence93 94preprocessed_input_exps = list(map(preprocess_input, input_exps))95preprocessed_target_exps = list(map(preprocess_target, target_exps))96 97preprocessed_input_exps[:5]98 99preprocessed_target_exps[:5]100 101def tokenize(lang):102 '''103 Tokenize the given list of strings and return the tokenized output104 along with the fitted tokenizer.105 '''106 lang_tokenizer = tf.keras.preprocessing.text.Tokenizer(filters='')107 lang_tokenizer.fit_on_texts(lang)108 tensor = lang_tokenizer.texts_to_sequences(lang)109 return tensor, lang_tokenizer110 111input_tensor, inp_lang_tokenizer = tokenize(preprocessed_input_exps)112 113len(inp_lang_tokenizer.word_index)114 115target_tensor, targ_lang_tokenizer = tokenize(preprocessed_target_exps)116 117old_len = len(targ_lang_tokenizer.word_index)118 119def append_start_end(x,last_int):120 '''121 Add integers for start and end tokens for input/target exps122 '''123 l = []124 l.append(last_int+1)125 l.extend(x)126 l.append(last_int+2)127 return l128 129input_tensor_list = [append_start_end(i,len(inp_lang_tokenizer.word_index)) for i in input_tensor]130target_tensor_list = [append_start_end(i,len(targ_lang_tokenizer.word_index)) for i in target_tensor]131 132# Pad all sequences such that they are of equal length133input_tensor = tf.keras.preprocessing.sequence.pad_sequences(input_tensor_list, padding='post')134target_tensor = tf.keras.preprocessing.sequence.pad_sequences(target_tensor_list, padding='post')135 136input_tensor137 138target_tensor139 140# Here we are increasing the vocabulary size of the target, by adding a141# few extra vocabulary words (which will not actually be used) as otherwise the142# small vocab size causes issues downstream in the network.143keys = [str(i) for i in range(10,51)]144for i,k in enumerate(keys):145 targ_lang_tokenizer.word_index[k]=len(targ_lang_tokenizer.word_index)+i+4146 147len(targ_lang_tokenizer.word_index)148 149# Creating training and validation sets150input_tensor_train, input_tensor_val, target_tensor_train, target_tensor_val = train_test_split(input_tensor,151 target_tensor,152 test_size=0.05,153 random_state=42)154 155len(input_tensor_train)156 157len(input_tensor_val)158 159BUFFER_SIZE = len(input_tensor_train)160BATCH_SIZE = 64161steps_per_epoch = len(input_tensor_train)//BATCH_SIZE162dataset = tf.data.Dataset.from_tensor_slices((input_tensor_train, target_tensor_train)).shuffle(BUFFER_SIZE)163dataset = dataset.batch(BATCH_SIZE, drop_remainder=True)164num_layers = 4165d_model = 128166dff = 512167num_heads = 8168input_vocab_size = len(inp_lang_tokenizer.word_index)+3169target_vocab_size = len(targ_lang_tokenizer.word_index)+3170dropout_rate = 0.0171 172example_input_batch, example_target_batch = next(iter(dataset))173example_input_batch.shape, example_target_batch.shape174 175# We provide positional information about the data to the model,176# otherwise each sentence will be treated as Bag of Words177def get_angles(pos, i, d_model):178 angle_rates = 1 / np.power(10000, (2 * (i//2)) / np.float32(d_model))179 return pos * angle_rates180 181def positional_encoding(position, d_model):182 angle_rads = get_angles(np.arange(position)[:, np.newaxis],183 np.arange(d_model)[np.newaxis, :],184 d_model)185 186 # apply sin to even indices in the array; 2i187 angle_rads[:, 0::2] = np.sin(angle_rads[:, 0::2])188 189 # apply cos to odd indices in the array; 2i+1190 angle_rads[:, 1::2] = np.cos(angle_rads[:, 1::2])191 192 pos_encoding = angle_rads[np.newaxis, ...]193 194 return tf.cast(pos_encoding, dtype=tf.float32)195 196# mask all elements are that not words (padding) so that it is not treated as input197def create_padding_mask(seq):198 seq = tf.cast(tf.math.equal(seq, 0), tf.float32)199 200 # add extra dimensions to add the padding201 # to the attention logits.202 return seq[:, tf.newaxis, tf.newaxis, :] # (batch_size, 1, 1, seq_len)203 204def create_look_ahead_mask(size):205 mask = 1 - tf.linalg.band_part(tf.ones((size, size)), -1, 0)206 return mask207 208dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)209 210def scaled_dot_product_attention(q, k, v, mask):211 matmul_qk = tf.matmul(q, k, transpose_b=True) # (..., seq_len_q, seq_len_k)212 213 # scale matmul_qk214 dk = tf.cast(tf.shape(k)[-1], tf.float32)215 scaled_attention_logits = matmul_qk / tf.math.sqrt(dk)216 217 # add the mask to the scaled tensor.218 if mask is not None:219 scaled_attention_logits += (mask * -1e9) 220 221 # softmax is normalized on the last axis (seq_len_k) so that the scores222 # add up to 1.223 attention_weights = tf.nn.softmax(scaled_attention_logits, axis=-1) # (..., seq_len_q, seq_len_k)224 225 output = tf.matmul(attention_weights, v) # (..., seq_len_q, depth_v)226 227 return output, attention_weights228 229class MultiHeadAttention(tf.keras.layers.Layer):230 def __init__(self, d_model, num_heads):231 super(MultiHeadAttention, self).__init__()232 self.num_heads = num_heads233 self.d_model = d_model234 235 assert d_model % self.num_heads == 0236 237 self.depth = d_model // self.num_heads238 239 self.wq = tf.keras.layers.Dense(d_model)240 self.wk = tf.keras.layers.Dense(d_model)241 self.wv = tf.keras.layers.Dense(d_model)242 243 self.dense = tf.keras.layers.Dense(d_model)244 245 def split_heads(self, x, batch_size):246 """Split the last dimension into (num_heads, depth).247 Transpose the result such that the shape is (batch_size, num_heads, seq_len, depth)248 """249 x = tf.reshape(x, (batch_size, -1, self.num_heads, self.depth))250 return tf.transpose(x, perm=[0, 2, 1, 3])251 252 def call(self, v, k, q, mask):253 batch_size = tf.shape(q)[0]254 255 q = self.wq(q) # (batch_size, seq_len, d_model)256 k = self.wk(k) # (batch_size, seq_len, d_model)257 v = self.wv(v) # (batch_size, seq_len, d_model)258 259 q = self.split_heads(q, batch_size) # (batch_size, num_heads, seq_len_q, depth)260 k = self.split_heads(k, batch_size) # (batch_size, num_heads, seq_len_k, depth)261 v = self.split_heads(v, batch_size) # (batch_size, num_heads, seq_len_v, depth)262 263 # scaled_attention.shape == (batch_size, num_heads, seq_len_q, depth)264 # attention_weights.shape == (batch_size, num_heads, seq_len_q, seq_len_k)265 scaled_attention, attention_weights = scaled_dot_product_attention(266 q, k, v, mask)267 268 scaled_attention = tf.transpose(scaled_attention, perm=[0, 2, 1, 3]) # (batch_size, seq_len_q, num_heads, depth)269 270 concat_attention = tf.reshape(scaled_attention, 271 (batch_size, -1, self.d_model)) # (batch_size, seq_len_q, d_model)272 273 output = self.dense(concat_attention) # (batch_size, seq_len_q, d_model)274 275 return output, attention_weights276 277def point_wise_feed_forward_network(d_model, dff):278 return tf.keras.Sequential([279 tf.keras.layers.Dense(dff, activation='relu'), # (batch_size, seq_len, dff)280 tf.keras.layers.Dense(d_model) # (batch_size, seq_len, d_model)281 ])282 283class EncoderLayer(tf.keras.layers.Layer):284 def __init__(self, d_model, num_heads, dff, rate=0.1):285 super(EncoderLayer, self).__init__()286 287 self.mha = MultiHeadAttention(d_model, num_heads)288 self.ffn = point_wise_feed_forward_network(d_model, dff)289 290 # normalize data per feature instead of batch291 self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)292 self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)293 294 self.dropout1 = tf.keras.layers.Dropout(rate)295 self.dropout2 = tf.keras.layers.Dropout(rate)296 297 def call(self, x, training, mask):298 # Multi-head attention layer299 attn_output, _ = self.mha(x, x, x, mask) 300 attn_output = self.dropout1(attn_output, training=training)301 # add residual connection to avoid vanishing gradient problem302 out1 = self.layernorm1(x + attn_output)303 304 # Feedforward layer305 ffn_output = self.ffn(out1)306 ffn_output = self.dropout2(ffn_output, training=training)307 # add residual connection to avoid vanishing gradient problem308 out2 = self.layernorm2(out1 + ffn_output)309 return out2310 311class Encoder(tf.keras.layers.Layer):312 def __init__(self, num_layers, d_model, num_heads, dff, input_vocab_size,313 maximum_position_encoding, rate=0.1):314 super(Encoder, self).__init__()315 316 self.d_model = d_model317 self.num_layers = num_layers318 319 self.embedding = tf.keras.layers.Embedding(input_vocab_size, d_model)320 self.pos_encoding = positional_encoding(maximum_position_encoding, 321 self.d_model)322 323 # Create encoder layers (count: num_layers)324 self.enc_layers = [EncoderLayer(d_model, num_heads, dff, rate) 325 for _ in range(num_layers)]326 327 self.dropout = tf.keras.layers.Dropout(rate)328 329 def call(self, x, training, mask):330 331 seq_len = tf.shape(x)[1]332 333 # adding embedding and position encoding.334 x = self.embedding(x) 335 x *= tf.math.sqrt(tf.cast(self.d_model, tf.float32))336 x += self.pos_encoding[:, :seq_len, :]337 338 x = self.dropout(x, training=training)339 340 for i in range(self.num_layers):341 x = self.enc_layers[i](x, training, mask)342 343 return x344 345class DecoderLayer(tf.keras.layers.Layer):346 def __init__(self, d_model, num_heads, dff, rate=0.1):347 super(DecoderLayer, self).__init__()348 349 self.mha1 = MultiHeadAttention(d_model, num_heads)350 self.mha2 = MultiHeadAttention(d_model, num_heads)351 352 self.ffn = point_wise_feed_forward_network(d_model, dff)353 354 self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)355 self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)356 self.layernorm3 = tf.keras.layers.LayerNormalization(epsilon=1e-6)357 358 self.dropout1 = tf.keras.layers.Dropout(rate)359 self.dropout2 = tf.keras.layers.Dropout(rate)360 self.dropout3 = tf.keras.layers.Dropout(rate)361 362 363 def call(self, x, enc_output, training, 364 look_ahead_mask, padding_mask):365 366 # Masked multihead attention layer (padding + look-ahead)367 attn1, attn_weights_block1 = self.mha1(x, x, x, look_ahead_mask)368 attn1 = self.dropout1(attn1, training=training)369 # again add residual connection370 out1 = self.layernorm1(attn1 + x)371 372 # Masked multihead attention layer (only padding)373 # with input from encoder as Key and Value, and input from previous layer as Query374 attn2, attn_weights_block2 = self.mha2(375 enc_output, enc_output, out1, padding_mask)376 attn2 = self.dropout2(attn2, training=training)377 # again add residual connection378 out2 = self.layernorm2(attn2 + out1)379 380 # Feedforward layer381 ffn_output = self.ffn(out2)382 ffn_output = self.dropout3(ffn_output, training=training)383 # again add residual connection384 out3 = self.layernorm3(ffn_output + out2)385 return out3, attn_weights_block1, attn_weights_block2386 387class Decoder(tf.keras.layers.Layer):388 def __init__(self, num_layers, d_model, num_heads, dff, target_vocab_size,389 maximum_position_encoding, rate=0.1):390 super(Decoder, self).__init__()391 392 self.d_model = d_model393 self.num_layers = num_layers394 395 self.embedding = tf.keras.layers.Embedding(target_vocab_size, d_model)396 self.pos_encoding = positional_encoding(maximum_position_encoding, d_model)397 398 # Create decoder layers (count: num_layers)399 self.dec_layers = [DecoderLayer(d_model, num_heads, dff, rate) 400 for _ in range(num_layers)]401 self.dropout = tf.keras.layers.Dropout(rate)402 403 def call(self, x, enc_output, training, 404 look_ahead_mask, padding_mask):405 406 seq_len = tf.shape(x)[1]407 attention_weights = {}408 409 x = self.embedding(x) # (batch_size, target_seq_len, d_model)410 411 x *= tf.math.sqrt(tf.cast(self.d_model, tf.float32))412 413 x += self.pos_encoding[:,:seq_len,:]414 415 x = self.dropout(x, training=training)416 417 for i in range(self.num_layers):418 x, block1, block2 = self.dec_layers[i](x, enc_output, training,419 look_ahead_mask, padding_mask)420 421 # store attenion weights, they can be used to visualize while translating422 attention_weights['decoder_layer{}_block1'.format(i+1)] = block1423 attention_weights['decoder_layer{}_block2'.format(i+1)] = block2424 425 return x, attention_weights426 427class Transformer(tf.keras.Model):428 def __init__(self, num_layers, d_model, num_heads, dff, input_vocab_size, 429 target_vocab_size, pe_input, pe_target, rate=0.1):430 super(Transformer, self).__init__()431 432 self.encoder = Encoder(num_layers, d_model, num_heads, dff, 433 input_vocab_size, pe_input, rate)434 435 self.decoder = Decoder(num_layers, d_model, num_heads, dff, 436 target_vocab_size, pe_target, rate)437 438 self.final_layer = tf.keras.layers.Dense(target_vocab_size)439 440 def call(self, inp, tar, training, enc_padding_mask, 441 look_ahead_mask, dec_padding_mask):442 443 # Pass the input to the encoder444 enc_output = self.encoder(inp, training, enc_padding_mask)445 446 # Pass the encoder output to the decoder447 dec_output, attention_weights = self.decoder(448 tar, enc_output, training, look_ahead_mask, dec_padding_mask)449 450 # Pass the decoder output to the last linear layer451 final_output = self.final_layer(dec_output)452 453 return final_output, attention_weights454 455class CustomSchedule(tf.keras.optimizers.schedules.LearningRateSchedule):456 def __init__(self, d_model, warmup_steps=4000):457 super(CustomSchedule, self).__init__()458 459 self.d_model = d_model460 self.d_model = tf.cast(self.d_model, tf.float32)461 462 self.warmup_steps = warmup_steps463 464 def __call__(self, step):465 arg1 = tf.math.rsqrt(step)466 arg2 = step * (self.warmup_steps ** -1.5)467 468 return tf.math.rsqrt(self.d_model) * tf.math.minimum(arg1, arg2)469 470learning_rate = CustomSchedule(d_model)471 472# Adam optimizer with a custom learning rate473optimizer = tf.keras.optimizers.Adam(learning_rate, beta_1=0.9, beta_2=0.98, 474 epsilon=1e-9)475 476loss_object = tf.keras.losses.SparseCategoricalCrossentropy(477 from_logits=True, reduction='none')478 479def loss_function(real, pred):480 # Apply a mask to paddings (0)481 mask = tf.math.logical_not(tf.math.equal(real, 0))482 loss_ = loss_object(real, pred)483 484 mask = tf.cast(mask, dtype=loss_.dtype)485 loss_ *= mask486 487 return tf.reduce_mean(loss_)488 489train_loss = tf.keras.metrics.Mean(name='train_loss')490train_accuracy = tf.keras.metrics.SparseCategoricalAccuracy(491 name='train_accuracy')492 493transformer = Transformer(num_layers, d_model, num_heads, dff,494 input_vocab_size, target_vocab_size, 495 pe_input=input_vocab_size, 496 pe_target=target_vocab_size,497 rate=dropout_rate)498 499def create_masks(inp, tar):500 # Encoder padding mask501 enc_padding_mask = create_padding_mask(inp)502 503 # Decoder padding mask504 dec_padding_mask = create_padding_mask(inp)505 506 # Look ahead mask (for hiding the rest of the sequence in the 1st decoder attention layer)507 look_ahead_mask = create_look_ahead_mask(tf.shape(tar)[1])508 dec_target_padding_mask = create_padding_mask(tar)509 combined_mask = tf.maximum(dec_target_padding_mask, look_ahead_mask)510 511 return enc_padding_mask, combined_mask, dec_padding_mask512 513# drive_root = '/gdrive/My Drive/'514drive_root = './'515 516checkpoint_dir = os.path.join(drive_root, "checkpoints")517checkpoint_dir = os.path.join(checkpoint_dir, "training_checkpoints/moops_transfomer")518 519print("Checkpoints directory is", checkpoint_dir)520if os.path.exists(checkpoint_dir):521 print("Checkpoints folder already exists")522else:523 print("Creating a checkpoints directory")524 os.makedirs(checkpoint_dir)525 526 527checkpoint = tf.train.Checkpoint(transformer=transformer,528 optimizer=optimizer)529 530ckpt_manager = tf.train.CheckpointManager(checkpoint, checkpoint_dir, max_to_keep=5)531 532latest = ckpt_manager.latest_checkpoint533latest534 535if latest:536 epoch_num = int(latest.split('/')[-1].split('-')[-1])537 checkpoint.restore(latest)538 print ('Latest checkpoint restored!!')539else:540 epoch_num = 0541 542epoch_num543 544# EPOCHS = 17545 546# def train_step(inp, tar):547# tar_inp = tar[:, :-1]548# tar_real = tar[:, 1:]549 550# enc_padding_mask, combined_mask, dec_padding_mask = create_masks(inp, tar_inp)551 552# with tf.GradientTape() as tape:553# predictions, _ = transformer(inp, tar_inp, 554# True, 555# enc_padding_mask, 556# combined_mask, 557# dec_padding_mask)558# loss = loss_function(tar_real, predictions)559 560# gradients = tape.gradient(loss, transformer.trainable_variables) 561# optimizer.apply_gradients(zip(gradients, transformer.trainable_variables))562 563# train_loss(loss)564# train_accuracy(tar_real, predictions)565 566# for epoch in range(epoch_num, EPOCHS):567# start = time.time()568 569# train_loss.reset_states()570# train_accuracy.reset_states()571 572# # inp -> question, tar -> equation573# for (batch, (inp, tar)) in enumerate(dataset):574# train_step(inp, tar)575 576# if batch % 50 == 0:577# print ('Epoch {} Batch {} Loss {:.4f} Accuracy {:.4f}'.format(578# epoch + 1, batch, train_loss.result(), train_accuracy.result()))579 580# ckpt_save_path = ckpt_manager.save()581# print ('Saving checkpoint for epoch {} at {}'.format(epoch+1,582# ckpt_save_path))583 584# print ('Epoch {} Loss {:.4f} Accuracy {:.4f}'.format(epoch + 1, 585# train_loss.result(), 586# train_accuracy.result()))587 588# print ('Time taken for 1 epoch: {} secs\n'.format(time.time() - start))589 590def evaluate(inp_sentence):591 start_token = [len(inp_lang_tokenizer.word_index)+1]592 end_token = [len(inp_lang_tokenizer.word_index)+2]593 594 # inp sentence is the word problem, hence adding the start and end token595 inp_sentence = start_token + [inp_lang_tokenizer.word_index.get(i, inp_lang_tokenizer.word_index['john']) for i in preprocess_input(inp_sentence).split(' ')] + end_token596 encoder_input = tf.expand_dims(inp_sentence, 0)597 598 # start with equation's start token599 decoder_input = [old_len+1]600 output = tf.expand_dims(decoder_input, 0)601 602 for i in range(MAX_LENGTH):603 enc_padding_mask, combined_mask, dec_padding_mask = create_masks(604 encoder_input, output)605 606 predictions, attention_weights = transformer(encoder_input, 607 output,608 False,609 enc_padding_mask,610 combined_mask,611 dec_padding_mask)612 613 # select the last word from the seq_len dimension614 predictions = predictions[: ,-1:, :] 615 predicted_id = tf.cast(tf.argmax(predictions, axis=-1), tf.int32)616 617 # return the result if the predicted_id is equal to the end token618 if predicted_id == old_len+2:619 return tf.squeeze(output, axis=0), attention_weights620 621 # concatentate the predicted_id to the output which is given to the decoder622 # as its input.623 output = tf.concat([output, predicted_id], axis=-1)624 return tf.squeeze(output, axis=0), attention_weights625 626# def plot_attention_weights(attention, sentence, result, layer):627# fig = plt.figure(figsize=(16, 8))628 629# sentence = preprocess_input(sentence)630 631# attention = tf.squeeze(attention[layer], axis=0)632 633# for head in range(attention.shape[0]):634# ax = fig.add_subplot(2, 4, head+1)635 636# # plot the attention weights637# ax.matshow(attention[head][:-1, :], cmap='viridis')638 639# fontdict = {'fontsize': 10}640 641# ax.set_xticks(range(len(sentence.split(' '))+2))642# ax.set_yticks(range(len([targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) 643# if i < len(targ_lang_tokenizer.word_index) and i not in [0,old_len+1,old_len+2]])+3))644 645 646# ax.set_ylim(len([targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) 647# if i < len(targ_lang_tokenizer.word_index) and i not in [0,old_len+1,old_len+2]]), -0.5)648 649# ax.set_xticklabels(650# ['<start>']+sentence.split(' ')+['<end>'], 651# fontdict=fontdict, rotation=90)652 653# ax.set_yticklabels([targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) 654# if i < len(targ_lang_tokenizer.word_index) and i not in [0,old_len+1,old_len+2]], 655# fontdict=fontdict)656 657# ax.set_xlabel('Head {}'.format(head+1))658 659# plt.tight_layout()660# plt.show()661 662MAX_LENGTH = 40663 664def translate(sentence, plot=''):665 666 667 668 result, attention_weights = evaluate(sentence)669 670 # use the result tokens to convert prediction into a list of characters671 # (not inclusing padding, start and end tokens)672 predicted_sentence = [targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) if (i < len(targ_lang_tokenizer.word_index) and i not in [0,46,47])] 673 674# print('Input: {}'.format(sentence))675 return ''.join(predicted_sentence)676 677 if plot:678 plot_attention_weights(attention_weights, sentence, result, plot)679 680# def evaluate_results(inp_sentence):681# start_token = [len(inp_lang_tokenizer.word_index)+1]682# end_token = [len(inp_lang_tokenizer.word_index)+2]683 684# # inp sentence is the word problem, hence adding the start and end token685# inp_sentence = start_token + list(inp_sentence.numpy()[0]) + end_token686 687# encoder_input = tf.expand_dims(inp_sentence, 0)688 689 690# decoder_input = [old_len+1]691# output = tf.expand_dims(decoder_input, 0)692 693# for i in range(MAX_LENGTH):694# enc_padding_mask, combined_mask, dec_padding_mask = create_masks(695# encoder_input, output)696 697# # predictions.shape == (batch_size, seq_len, vocab_size)698# predictions, attention_weights = transformer(encoder_input, 699# output,700# False,701# enc_padding_mask,702# combined_mask,703# dec_padding_mask)704 705# # select the last word from the seq_len dimension706# predictions = predictions[: ,-1:, :] # (batch_size, 1, vocab_size)707 708# predicted_id = tf.cast(tf.argmax(predictions, axis=-1), tf.int32)709 710# # return the result if the predicted_id is equal to the end token711# if predicted_id == old_len+2:712# return tf.squeeze(output, axis=0), attention_weights713 714# # concatentate the predicted_id to the output which is given to the decoder715# # as its input.716# output = tf.concat([output, predicted_id], axis=-1)717 718# return tf.squeeze(output, axis=0), attention_weights719 720# dataset_val = tf.data.Dataset.from_tensor_slices((input_tensor_val, target_tensor_val)).shuffle(BUFFER_SIZE)721# dataset_val = dataset_val.batch(1, drop_remainder=True)722 723# y_true = []724# y_pred = []725# acc_cnt = 0726 727# a = 0728# for (inp_val_batch, target_val_batch) in iter(dataset_val):729# a += 1730# if a % 100 == 0:731# print(a)732# print("Accuracy count: ",acc_cnt)733# print('------------------')734# target_sentence = ''735# for i in target_val_batch.numpy()[0]:736# if i not in [0,old_len+1,old_len+2]:737# target_sentence += (targ_lang_tokenizer.index_word[i] + ' ')738 739# y_true.append([target_sentence.split(' ')[:-1]])740 741# result, _ = evaluate_results(inp_val_batch)742# predicted_sentence = [targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) if (i < len(targ_lang_tokenizer.word_index) and i not in [0,old_len+1,old_len+2])] 743# y_pred.append(predicted_sentence)744 745# if target_sentence.split(' ')[:-1] == predicted_sentence:746# acc_cnt += 1747 748# len(y_true), len(y_pred)749 750# print('Corpus BLEU score of the model: ', corpus_bleu(y_true, y_pred))751 752# print('Accuracy of the model: ', acc_cnt/len(input_tensor_val))753 754check_str = ' '.join([inp_lang_tokenizer.index_word[i] for i in input_tensor_val[242] if i not in [0,755 len(inp_lang_tokenizer.word_index)+1,756 len(inp_lang_tokenizer.word_index)+2]])757 758check_str759 760translate(check_str)761 762#'victor had some car . john took 3 0 from him . now victor has 6 8 car . how many car victor had originally ?'763translate('Nafis had 31 raspberry . He slice each raspberry into 19 slices . How many raspberry slices did Denise make?')764 765interface = gr.Interface(766 fn = translate,767 inputs = gr.inputs.Textbox(lines = 2), 768 outputs = 'text',769 examples = [770 ['Rachel bought two coloring books. One had 23 pictures and the other had 32. After one week she had colored 19 of the pictures. How many pictures does she still have to color?'],771 ['Denise had 31 raspberries. He slices each raspberry into 19 slices. How many raspberry slices did Denise make?'],772 ['A painter needed to paint 12 rooms in a building. Each room takes 7 hours to paint. If he already painted 5 rooms, how much longer will he take to paint the rest?'],773 ['Jerry had 135 pens. John took 19 pens from him. How many pens Jerry have left?'],774 ['Donald had some apples. Hillary took 20 apples from him. Now Donald has 100 apples. How many apples Donald had before?']775 ],776 title = 'Mathbot',777 description = 'Enter a simple math word problem and our AI will try to predict an expression to solve it. Mathbot occasionally makes mistakes. Feel free to press "flag" if you encounter such a scenario.',778 )779interface.launch()