Team Ai
Apppublic

Casio991ms/MathBot

sourceHugging Facemitupdated 4y agoView on Hugging Face
6likes
app.py779 linesDownload Raw Back to root
1# -*- coding: utf-8 -*-2"""MWP_Solver_-_Transformer_with_Multi-head_Attention_Block (1).ipynb3 4Automatically generated by Colaboratory.5 6Original file is located at7    https://colab.research.google.com/drive/1Tn_j0k8EJ7ny_h7Pjm0stJhNMG4si_y_8"""9 10# ! pip install -q gradio11 12import pandas as pd13import re14import os15import time16import random17import numpy as np18 19os.system("pip install tensorflow")20os.system("pip install scikit-learn")21os.system("pip install spacy")22os.system("pip install nltk")23os.system("spacy download en_core_web_sm")24 25import tensorflow as tf26import matplotlib.pyplot as plt27import matplotlib.ticker as ticker28from sklearn.model_selection import train_test_split29 30import pickle31 32import spacy33 34from nltk.translate.bleu_score import corpus_bleu35 36import gradio as gr37 38os.system("wget -nc 'https://docs.google.com/uc?export=download&id=1Y8Ee4lUs30BAfFtL3d3VjwChmbDG7O6H' -O data_final.pkl")39os.system('''wget --load-cookies /tmp/cookies.txt "https://docs.google.com/uc?export=download&confirm=$(wget --quiet --save-cookies /tmp/cookies.txt --keep-session-cookies --no-check-certificate 'https://docs.google.com/uc?export=download&id=1gAQVaxg_2mNcr8qwx0J2UwpkvoJgLu6a' -O- | sed -rn 's/.*confirm=([0-9A-Za-z_]+).*/\\1\\n/p')&id=1gAQVaxg_2mNcr8qwx0J2UwpkvoJgLu6a" -O checkpoints.zip && rm -rf /tmp/cookies.txt''')40os.system("unzip -n './checkpoints.zip' -d './'")41 42nlp = spacy.load("en_core_web_sm")43 44tf.__version__45 46with open('data_final.pkl', 'rb') as f:47  df = pickle.load(f)48 49df.shape50 51df.head()52 53input_exps = list(df['Question'].values)54 55def convert_eqn(eqn):56  '''57  Add a space between every character in the equation string.58  Eg: 'x = 23 + 88' becomes 'x =  2 3 + 8 8'59  '''60  elements = list(eqn)61  return ' '.join(elements)62 63target_exps = list(df['Equation'].apply(lambda x: convert_eqn(x)).values)64 65# Input: Word problem66input_exps[:5]67 68# Target: Equation69target_exps[:5]70 71len(pd.Series(input_exps)), len(pd.Series(input_exps).unique())72 73len(pd.Series(target_exps)), len(pd.Series(target_exps).unique())74 75def preprocess_input(sentence):76  '''77  For the word problem, convert everything to lowercase, add spaces around all78  punctuations and digits, and remove any extra spaces. 79  '''80  sentence = sentence.lower().strip()81  sentence = re.sub(r"([?.!,’])", r" \1 ", sentence)82  sentence = re.sub(r"([0-9])", r" \1 ", sentence)83  sentence = re.sub(r'[" "]+', " ", sentence)84  sentence = sentence.rstrip().strip()85  return sentence86 87def preprocess_target(sentence):88  '''89  For the equation, convert it to lowercase and remove extra spaces90  '''91  sentence = sentence.lower().strip()92  return sentence93 94preprocessed_input_exps = list(map(preprocess_input, input_exps))95preprocessed_target_exps = list(map(preprocess_target, target_exps))96 97preprocessed_input_exps[:5]98 99preprocessed_target_exps[:5]100 101def tokenize(lang):102  '''103  Tokenize the given list of strings and return the tokenized output104  along with the fitted tokenizer.105  '''106  lang_tokenizer = tf.keras.preprocessing.text.Tokenizer(filters='')107  lang_tokenizer.fit_on_texts(lang)108  tensor = lang_tokenizer.texts_to_sequences(lang)109  return tensor, lang_tokenizer110 111input_tensor, inp_lang_tokenizer = tokenize(preprocessed_input_exps)112 113len(inp_lang_tokenizer.word_index)114 115target_tensor, targ_lang_tokenizer = tokenize(preprocessed_target_exps)116 117old_len = len(targ_lang_tokenizer.word_index)118 119def append_start_end(x,last_int):120  '''121  Add integers for start and end tokens for input/target exps122  '''123  l = []124  l.append(last_int+1)125  l.extend(x)126  l.append(last_int+2)127  return l128 129input_tensor_list = [append_start_end(i,len(inp_lang_tokenizer.word_index)) for i in input_tensor]130target_tensor_list = [append_start_end(i,len(targ_lang_tokenizer.word_index)) for i in target_tensor]131 132# Pad all sequences such that they are of equal length133input_tensor = tf.keras.preprocessing.sequence.pad_sequences(input_tensor_list, padding='post')134target_tensor = tf.keras.preprocessing.sequence.pad_sequences(target_tensor_list, padding='post')135 136input_tensor137 138target_tensor139 140# Here we are increasing the vocabulary size of the target, by adding a141# few extra vocabulary words (which will not actually be used) as otherwise the142# small vocab size causes issues downstream in the network.143keys = [str(i) for i in range(10,51)]144for i,k in enumerate(keys):145  targ_lang_tokenizer.word_index[k]=len(targ_lang_tokenizer.word_index)+i+4146 147len(targ_lang_tokenizer.word_index)148 149# Creating training and validation sets150input_tensor_train, input_tensor_val, target_tensor_train, target_tensor_val = train_test_split(input_tensor,151                                                                                                target_tensor,152                                                                                                test_size=0.05,153                                                                                                random_state=42)154 155len(input_tensor_train)156 157len(input_tensor_val)158 159BUFFER_SIZE = len(input_tensor_train)160BATCH_SIZE = 64161steps_per_epoch = len(input_tensor_train)//BATCH_SIZE162dataset = tf.data.Dataset.from_tensor_slices((input_tensor_train, target_tensor_train)).shuffle(BUFFER_SIZE)163dataset = dataset.batch(BATCH_SIZE, drop_remainder=True)164num_layers = 4165d_model = 128166dff = 512167num_heads = 8168input_vocab_size = len(inp_lang_tokenizer.word_index)+3169target_vocab_size = len(targ_lang_tokenizer.word_index)+3170dropout_rate = 0.0171 172example_input_batch, example_target_batch = next(iter(dataset))173example_input_batch.shape, example_target_batch.shape174 175# We provide positional information about the data to the model,176# otherwise each sentence will be treated as Bag of Words177def get_angles(pos, i, d_model):178  angle_rates = 1 / np.power(10000, (2 * (i//2)) / np.float32(d_model))179  return pos * angle_rates180 181def positional_encoding(position, d_model):182  angle_rads = get_angles(np.arange(position)[:, np.newaxis],183                          np.arange(d_model)[np.newaxis, :],184                          d_model)185  186  # apply sin to even indices in the array; 2i187  angle_rads[:, 0::2] = np.sin(angle_rads[:, 0::2])188  189  # apply cos to odd indices in the array; 2i+1190  angle_rads[:, 1::2] = np.cos(angle_rads[:, 1::2])191    192  pos_encoding = angle_rads[np.newaxis, ...]193    194  return tf.cast(pos_encoding, dtype=tf.float32)195 196# mask all elements are that not words (padding) so that it is not treated as input197def create_padding_mask(seq):198  seq = tf.cast(tf.math.equal(seq, 0), tf.float32)199  200  # add extra dimensions to add the padding201  # to the attention logits.202  return seq[:, tf.newaxis, tf.newaxis, :]  # (batch_size, 1, 1, seq_len)203 204def create_look_ahead_mask(size):205  mask = 1 - tf.linalg.band_part(tf.ones((size, size)), -1, 0)206  return mask207 208dataset = dataset.prefetch(tf.data.experimental.AUTOTUNE)209 210def scaled_dot_product_attention(q, k, v, mask):211  matmul_qk = tf.matmul(q, k, transpose_b=True)  # (..., seq_len_q, seq_len_k)212  213  # scale matmul_qk214  dk = tf.cast(tf.shape(k)[-1], tf.float32)215  scaled_attention_logits = matmul_qk / tf.math.sqrt(dk)216 217  # add the mask to the scaled tensor.218  if mask is not None:219    scaled_attention_logits += (mask * -1e9)  220 221  # softmax is normalized on the last axis (seq_len_k) so that the scores222  # add up to 1.223  attention_weights = tf.nn.softmax(scaled_attention_logits, axis=-1)  # (..., seq_len_q, seq_len_k)224 225  output = tf.matmul(attention_weights, v)  # (..., seq_len_q, depth_v)226 227  return output, attention_weights228 229class MultiHeadAttention(tf.keras.layers.Layer):230  def __init__(self, d_model, num_heads):231    super(MultiHeadAttention, self).__init__()232    self.num_heads = num_heads233    self.d_model = d_model234    235    assert d_model % self.num_heads == 0236    237    self.depth = d_model // self.num_heads238    239    self.wq = tf.keras.layers.Dense(d_model)240    self.wk = tf.keras.layers.Dense(d_model)241    self.wv = tf.keras.layers.Dense(d_model)242    243    self.dense = tf.keras.layers.Dense(d_model)244        245  def split_heads(self, x, batch_size):246    """Split the last dimension into (num_heads, depth).247    Transpose the result such that the shape is (batch_size, num_heads, seq_len, depth)248    """249    x = tf.reshape(x, (batch_size, -1, self.num_heads, self.depth))250    return tf.transpose(x, perm=[0, 2, 1, 3])251    252  def call(self, v, k, q, mask):253    batch_size = tf.shape(q)[0]254    255    q = self.wq(q)  # (batch_size, seq_len, d_model)256    k = self.wk(k)  # (batch_size, seq_len, d_model)257    v = self.wv(v)  # (batch_size, seq_len, d_model)258    259    q = self.split_heads(q, batch_size)  # (batch_size, num_heads, seq_len_q, depth)260    k = self.split_heads(k, batch_size)  # (batch_size, num_heads, seq_len_k, depth)261    v = self.split_heads(v, batch_size)  # (batch_size, num_heads, seq_len_v, depth)262    263    # scaled_attention.shape == (batch_size, num_heads, seq_len_q, depth)264    # attention_weights.shape == (batch_size, num_heads, seq_len_q, seq_len_k)265    scaled_attention, attention_weights = scaled_dot_product_attention(266        q, k, v, mask)267    268    scaled_attention = tf.transpose(scaled_attention, perm=[0, 2, 1, 3])  # (batch_size, seq_len_q, num_heads, depth)269 270    concat_attention = tf.reshape(scaled_attention, 271                                  (batch_size, -1, self.d_model))  # (batch_size, seq_len_q, d_model)272 273    output = self.dense(concat_attention)  # (batch_size, seq_len_q, d_model)274        275    return output, attention_weights276 277def point_wise_feed_forward_network(d_model, dff):278  return tf.keras.Sequential([279      tf.keras.layers.Dense(dff, activation='relu'),  # (batch_size, seq_len, dff)280      tf.keras.layers.Dense(d_model)  # (batch_size, seq_len, d_model)281  ])282 283class EncoderLayer(tf.keras.layers.Layer):284  def __init__(self, d_model, num_heads, dff, rate=0.1):285    super(EncoderLayer, self).__init__()286 287    self.mha = MultiHeadAttention(d_model, num_heads)288    self.ffn = point_wise_feed_forward_network(d_model, dff)289 290    # normalize data per feature instead of batch291    self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)292    self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)293    294    self.dropout1 = tf.keras.layers.Dropout(rate)295    self.dropout2 = tf.keras.layers.Dropout(rate)296    297  def call(self, x, training, mask):298    # Multi-head attention layer299    attn_output, _ = self.mha(x, x, x, mask) 300    attn_output = self.dropout1(attn_output, training=training)301    # add residual connection to avoid vanishing gradient problem302    out1 = self.layernorm1(x + attn_output)303    304    # Feedforward layer305    ffn_output = self.ffn(out1)306    ffn_output = self.dropout2(ffn_output, training=training)307    # add residual connection to avoid vanishing gradient problem308    out2 = self.layernorm2(out1 + ffn_output)309    return out2310 311class Encoder(tf.keras.layers.Layer):312  def __init__(self, num_layers, d_model, num_heads, dff, input_vocab_size,313               maximum_position_encoding, rate=0.1):314    super(Encoder, self).__init__()315 316    self.d_model = d_model317    self.num_layers = num_layers318    319    self.embedding = tf.keras.layers.Embedding(input_vocab_size, d_model)320    self.pos_encoding = positional_encoding(maximum_position_encoding, 321                                            self.d_model)322    323    # Create encoder layers (count: num_layers)324    self.enc_layers = [EncoderLayer(d_model, num_heads, dff, rate) 325                       for _ in range(num_layers)]326  327    self.dropout = tf.keras.layers.Dropout(rate)328        329  def call(self, x, training, mask):330 331    seq_len = tf.shape(x)[1]332 333    # adding embedding and position encoding.334    x = self.embedding(x)  335    x *= tf.math.sqrt(tf.cast(self.d_model, tf.float32))336    x += self.pos_encoding[:, :seq_len, :]337 338    x = self.dropout(x, training=training)339    340    for i in range(self.num_layers):341      x = self.enc_layers[i](x, training, mask)342    343    return x344 345class DecoderLayer(tf.keras.layers.Layer):346  def __init__(self, d_model, num_heads, dff, rate=0.1):347    super(DecoderLayer, self).__init__()348 349    self.mha1 = MultiHeadAttention(d_model, num_heads)350    self.mha2 = MultiHeadAttention(d_model, num_heads)351 352    self.ffn = point_wise_feed_forward_network(d_model, dff)353 354    self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)355    self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)356    self.layernorm3 = tf.keras.layers.LayerNormalization(epsilon=1e-6)357    358    self.dropout1 = tf.keras.layers.Dropout(rate)359    self.dropout2 = tf.keras.layers.Dropout(rate)360    self.dropout3 = tf.keras.layers.Dropout(rate)361    362    363  def call(self, x, enc_output, training, 364           look_ahead_mask, padding_mask):365 366    # Masked multihead attention layer (padding + look-ahead)367    attn1, attn_weights_block1 = self.mha1(x, x, x, look_ahead_mask)368    attn1 = self.dropout1(attn1, training=training)369    # again add residual connection370    out1 = self.layernorm1(attn1 + x)371    372    # Masked multihead attention layer (only padding)373    # with input from encoder as Key and Value, and input from previous layer as Query374    attn2, attn_weights_block2 = self.mha2(375        enc_output, enc_output, out1, padding_mask)376    attn2 = self.dropout2(attn2, training=training)377    # again add residual connection378    out2 = self.layernorm2(attn2 + out1)379    380    # Feedforward layer381    ffn_output = self.ffn(out2)382    ffn_output = self.dropout3(ffn_output, training=training)383    # again add residual connection384    out3 = self.layernorm3(ffn_output + out2)385    return out3, attn_weights_block1, attn_weights_block2386 387class Decoder(tf.keras.layers.Layer):388  def __init__(self, num_layers, d_model, num_heads, dff, target_vocab_size,389               maximum_position_encoding, rate=0.1):390    super(Decoder, self).__init__()391 392    self.d_model = d_model393    self.num_layers = num_layers394     395    self.embedding = tf.keras.layers.Embedding(target_vocab_size, d_model)396    self.pos_encoding = positional_encoding(maximum_position_encoding, d_model)397    398    # Create decoder layers (count: num_layers)399    self.dec_layers = [DecoderLayer(d_model, num_heads, dff, rate) 400                       for _ in range(num_layers)]401    self.dropout = tf.keras.layers.Dropout(rate)402    403  def call(self, x, enc_output, training, 404           look_ahead_mask, padding_mask):405 406    seq_len = tf.shape(x)[1]407    attention_weights = {}408    409    x = self.embedding(x)  # (batch_size, target_seq_len, d_model)410    411    x *= tf.math.sqrt(tf.cast(self.d_model, tf.float32))412    413    x += self.pos_encoding[:,:seq_len,:]414    415    x = self.dropout(x, training=training)416 417    for i in range(self.num_layers):418      x, block1, block2 = self.dec_layers[i](x, enc_output, training,419                                             look_ahead_mask, padding_mask)420      421      # store attenion weights, they can be used to visualize while translating422      attention_weights['decoder_layer{}_block1'.format(i+1)] = block1423      attention_weights['decoder_layer{}_block2'.format(i+1)] = block2424    425    return x, attention_weights426 427class Transformer(tf.keras.Model):428  def __init__(self, num_layers, d_model, num_heads, dff, input_vocab_size, 429               target_vocab_size, pe_input, pe_target, rate=0.1):430    super(Transformer, self).__init__()431 432    self.encoder = Encoder(num_layers, d_model, num_heads, dff, 433                           input_vocab_size, pe_input, rate)434 435    self.decoder = Decoder(num_layers, d_model, num_heads, dff, 436                           target_vocab_size, pe_target, rate)437 438    self.final_layer = tf.keras.layers.Dense(target_vocab_size)439    440  def call(self, inp, tar, training, enc_padding_mask, 441           look_ahead_mask, dec_padding_mask):442 443    # Pass the input to the encoder444    enc_output = self.encoder(inp, training, enc_padding_mask)445    446    # Pass the encoder output to the decoder447    dec_output, attention_weights = self.decoder(448        tar, enc_output, training, look_ahead_mask, dec_padding_mask)449    450    # Pass the decoder output to the last linear layer451    final_output = self.final_layer(dec_output)452    453    return final_output, attention_weights454 455class CustomSchedule(tf.keras.optimizers.schedules.LearningRateSchedule):456  def __init__(self, d_model, warmup_steps=4000):457    super(CustomSchedule, self).__init__()458    459    self.d_model = d_model460    self.d_model = tf.cast(self.d_model, tf.float32)461 462    self.warmup_steps = warmup_steps463    464  def __call__(self, step):465    arg1 = tf.math.rsqrt(step)466    arg2 = step * (self.warmup_steps ** -1.5)467    468    return tf.math.rsqrt(self.d_model) * tf.math.minimum(arg1, arg2)469 470learning_rate = CustomSchedule(d_model)471 472# Adam optimizer with a custom learning rate473optimizer = tf.keras.optimizers.Adam(learning_rate, beta_1=0.9, beta_2=0.98, 474                                     epsilon=1e-9)475 476loss_object = tf.keras.losses.SparseCategoricalCrossentropy(477    from_logits=True, reduction='none')478 479def loss_function(real, pred):480  # Apply a mask to paddings (0)481  mask = tf.math.logical_not(tf.math.equal(real, 0))482  loss_ = loss_object(real, pred)483 484  mask = tf.cast(mask, dtype=loss_.dtype)485  loss_ *= mask486  487  return tf.reduce_mean(loss_)488 489train_loss = tf.keras.metrics.Mean(name='train_loss')490train_accuracy = tf.keras.metrics.SparseCategoricalAccuracy(491    name='train_accuracy')492 493transformer = Transformer(num_layers, d_model, num_heads, dff,494                          input_vocab_size, target_vocab_size, 495                          pe_input=input_vocab_size, 496                          pe_target=target_vocab_size,497                          rate=dropout_rate)498 499def create_masks(inp, tar):500  # Encoder padding mask501  enc_padding_mask = create_padding_mask(inp)502  503  # Decoder padding mask504  dec_padding_mask = create_padding_mask(inp)505  506  # Look ahead mask (for hiding the rest of the sequence in the 1st decoder attention layer)507  look_ahead_mask = create_look_ahead_mask(tf.shape(tar)[1])508  dec_target_padding_mask = create_padding_mask(tar)509  combined_mask = tf.maximum(dec_target_padding_mask, look_ahead_mask)510  511  return enc_padding_mask, combined_mask, dec_padding_mask512 513# drive_root = '/gdrive/My Drive/'514drive_root = './'515 516checkpoint_dir = os.path.join(drive_root, "checkpoints")517checkpoint_dir = os.path.join(checkpoint_dir, "training_checkpoints/moops_transfomer")518 519print("Checkpoints directory is", checkpoint_dir)520if os.path.exists(checkpoint_dir):521  print("Checkpoints folder already exists")522else:523  print("Creating a checkpoints directory")524  os.makedirs(checkpoint_dir)525 526 527checkpoint = tf.train.Checkpoint(transformer=transformer,528                           optimizer=optimizer)529 530ckpt_manager = tf.train.CheckpointManager(checkpoint, checkpoint_dir, max_to_keep=5)531 532latest = ckpt_manager.latest_checkpoint533latest534 535if latest:536  epoch_num = int(latest.split('/')[-1].split('-')[-1])537  checkpoint.restore(latest)538  print ('Latest checkpoint restored!!')539else:540  epoch_num = 0541 542epoch_num543 544# EPOCHS = 17545 546# def train_step(inp, tar):547#   tar_inp = tar[:, :-1]548#   tar_real = tar[:, 1:]549  550#   enc_padding_mask, combined_mask, dec_padding_mask = create_masks(inp, tar_inp)551  552#   with tf.GradientTape() as tape:553#     predictions, _ = transformer(inp, tar_inp, 554#                                  True, 555#                                  enc_padding_mask, 556#                                  combined_mask, 557#                                  dec_padding_mask)558#     loss = loss_function(tar_real, predictions)559 560#   gradients = tape.gradient(loss, transformer.trainable_variables)    561#   optimizer.apply_gradients(zip(gradients, transformer.trainable_variables))562  563#   train_loss(loss)564#   train_accuracy(tar_real, predictions)565 566# for epoch in range(epoch_num, EPOCHS):567#   start = time.time()568  569#   train_loss.reset_states()570#   train_accuracy.reset_states()571  572#   # inp -> question, tar -> equation573#   for (batch, (inp, tar)) in enumerate(dataset):574#     train_step(inp, tar)575    576#     if batch % 50 == 0:577#       print ('Epoch {} Batch {} Loss {:.4f} Accuracy {:.4f}'.format(578#           epoch + 1, batch, train_loss.result(), train_accuracy.result()))579      580#   ckpt_save_path = ckpt_manager.save()581#   print ('Saving checkpoint for epoch {} at {}'.format(epoch+1,582#                                                          ckpt_save_path))583    584#   print ('Epoch {} Loss {:.4f} Accuracy {:.4f}'.format(epoch + 1, 585#                                                 train_loss.result(), 586#                                                 train_accuracy.result()))587 588#   print ('Time taken for 1 epoch: {} secs\n'.format(time.time() - start))589 590def evaluate(inp_sentence):591  start_token = [len(inp_lang_tokenizer.word_index)+1]592  end_token = [len(inp_lang_tokenizer.word_index)+2]593  594  # inp sentence is the word problem, hence adding the start and end token595  inp_sentence = start_token + [inp_lang_tokenizer.word_index.get(i, inp_lang_tokenizer.word_index['john']) for i in preprocess_input(inp_sentence).split(' ')] + end_token596  encoder_input = tf.expand_dims(inp_sentence, 0)597  598  # start with equation's start token599  decoder_input = [old_len+1]600  output = tf.expand_dims(decoder_input, 0)601    602  for i in range(MAX_LENGTH):603    enc_padding_mask, combined_mask, dec_padding_mask = create_masks(604        encoder_input, output)605  606    predictions, attention_weights = transformer(encoder_input, 607                                                 output,608                                                 False,609                                                 enc_padding_mask,610                                                 combined_mask,611                                                 dec_padding_mask)612    613    # select the last word from the seq_len dimension614    predictions = predictions[: ,-1:, :] 615    predicted_id = tf.cast(tf.argmax(predictions, axis=-1), tf.int32)616    617    # return the result if the predicted_id is equal to the end token618    if predicted_id == old_len+2:619      return tf.squeeze(output, axis=0), attention_weights620    621    # concatentate the predicted_id to the output which is given to the decoder622    # as its input.623    output = tf.concat([output, predicted_id], axis=-1)624  return tf.squeeze(output, axis=0), attention_weights625 626# def plot_attention_weights(attention, sentence, result, layer):627#   fig = plt.figure(figsize=(16, 8))628  629#   sentence = preprocess_input(sentence)630  631#   attention = tf.squeeze(attention[layer], axis=0)632  633#   for head in range(attention.shape[0]):634#     ax = fig.add_subplot(2, 4, head+1)635    636#     # plot the attention weights637#     ax.matshow(attention[head][:-1, :], cmap='viridis')638    639#     fontdict = {'fontsize': 10}640    641#     ax.set_xticks(range(len(sentence.split(' '))+2))642#     ax.set_yticks(range(len([targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) 643#                         if i < len(targ_lang_tokenizer.word_index) and i not in [0,old_len+1,old_len+2]])+3))644    645    646#     ax.set_ylim(len([targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) 647#                         if i < len(targ_lang_tokenizer.word_index) and i not in [0,old_len+1,old_len+2]]), -0.5)648        649#     ax.set_xticklabels(650#         ['<start>']+sentence.split(' ')+['<end>'], 651#         fontdict=fontdict, rotation=90)652    653#     ax.set_yticklabels([targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) 654#                         if i < len(targ_lang_tokenizer.word_index) and i not in [0,old_len+1,old_len+2]], 655#                        fontdict=fontdict)656    657#     ax.set_xlabel('Head {}'.format(head+1))658  659#   plt.tight_layout()660#   plt.show()661 662MAX_LENGTH = 40663 664def translate(sentence, plot=''):665 666    667 668    result, attention_weights = evaluate(sentence)669 670    # use the result tokens to convert prediction into a list of characters671    # (not inclusing padding, start and end tokens)672    predicted_sentence = [targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) if (i < len(targ_lang_tokenizer.word_index) and i not in [0,46,47])]  673 674#   print('Input: {}'.format(sentence))675    return ''.join(predicted_sentence)676  677    if plot:678        plot_attention_weights(attention_weights, sentence, result, plot)679 680# def evaluate_results(inp_sentence):681#   start_token = [len(inp_lang_tokenizer.word_index)+1]682#   end_token = [len(inp_lang_tokenizer.word_index)+2]683  684#   # inp sentence is the word problem, hence adding the start and end token685#   inp_sentence = start_token + list(inp_sentence.numpy()[0]) + end_token686  687#   encoder_input = tf.expand_dims(inp_sentence, 0)688  689  690#   decoder_input = [old_len+1]691#   output = tf.expand_dims(decoder_input, 0)692    693#   for i in range(MAX_LENGTH):694#     enc_padding_mask, combined_mask, dec_padding_mask = create_masks(695#         encoder_input, output)696  697#     # predictions.shape == (batch_size, seq_len, vocab_size)698#     predictions, attention_weights = transformer(encoder_input, 699#                                                  output,700#                                                  False,701#                                                  enc_padding_mask,702#                                                  combined_mask,703#                                                  dec_padding_mask)704    705#     # select the last word from the seq_len dimension706#     predictions = predictions[: ,-1:, :]  # (batch_size, 1, vocab_size)707 708#     predicted_id = tf.cast(tf.argmax(predictions, axis=-1), tf.int32)709    710#     # return the result if the predicted_id is equal to the end token711#     if predicted_id == old_len+2:712#       return tf.squeeze(output, axis=0), attention_weights713    714#     # concatentate the predicted_id to the output which is given to the decoder715#     # as its input.716#     output = tf.concat([output, predicted_id], axis=-1)717 718#   return tf.squeeze(output, axis=0), attention_weights719 720# dataset_val = tf.data.Dataset.from_tensor_slices((input_tensor_val, target_tensor_val)).shuffle(BUFFER_SIZE)721# dataset_val = dataset_val.batch(1, drop_remainder=True)722 723# y_true = []724# y_pred = []725# acc_cnt = 0726 727# a = 0728# for (inp_val_batch, target_val_batch) in iter(dataset_val):729#   a += 1730#   if a % 100 == 0:731#     print(a)732#     print("Accuracy count: ",acc_cnt)733#     print('------------------')734#   target_sentence = ''735#   for i in target_val_batch.numpy()[0]:736#     if i not in [0,old_len+1,old_len+2]:737#       target_sentence += (targ_lang_tokenizer.index_word[i] + ' ')738  739#   y_true.append([target_sentence.split(' ')[:-1]])740  741#   result, _ = evaluate_results(inp_val_batch)742#   predicted_sentence = [targ_lang_tokenizer.index_word[i] for i in list(result.numpy()) if (i < len(targ_lang_tokenizer.word_index) and i not in [0,old_len+1,old_len+2])] 743#   y_pred.append(predicted_sentence)744  745#   if target_sentence.split(' ')[:-1] == predicted_sentence:746#     acc_cnt += 1747 748# len(y_true), len(y_pred)749 750# print('Corpus BLEU score of the model: ', corpus_bleu(y_true, y_pred))751 752# print('Accuracy of the model: ', acc_cnt/len(input_tensor_val))753 754check_str = ' '.join([inp_lang_tokenizer.index_word[i] for i in input_tensor_val[242] if i not in [0,755                                                                                                  len(inp_lang_tokenizer.word_index)+1,756                                                                                                  len(inp_lang_tokenizer.word_index)+2]])757 758check_str759 760translate(check_str)761 762#'victor had some car . john took 3 0 from him . now victor has 6 8 car . how many car victor had originally ?'763translate('Nafis had 31 raspberry . He slice each raspberry into 19 slices . How many raspberry slices did Denise make?')764 765interface = gr.Interface(766    fn = translate,767    inputs = gr.inputs.Textbox(lines = 2), 768    outputs = 'text',769    examples = [770                ['Rachel bought two coloring books. One had 23 pictures and the other had 32. After one week she had colored 19 of the pictures. How many pictures does she still have to color?'],771                ['Denise had 31 raspberries. He slices each raspberry into 19 slices. How many raspberry slices did Denise make?'],772                ['A painter needed to paint 12 rooms in a building. Each room takes 7 hours to paint. If he already painted 5 rooms, how much longer will he take to paint the rest?'],773                ['Jerry had 135 pens. John took 19 pens from him. How many pens Jerry have left?'],774                ['Donald had some apples. Hillary took 20 apples from him. Now Donald has 100 apples. How many apples Donald had before?']775    ],776    title = 'Mathbot',777    description = 'Enter a simple math word problem and our AI will try to predict an expression to solve it. Mathbot occasionally makes mistakes. Feel free to press "flag" if you encounter such a scenario.',778    )779interface.launch()