__Vanilla_Tfn__.tensorflow.py

import time

class ElapsedTimer(object):
    def __init__(self):
        self.start_time = time.time()
    def elapsed(self,sec):
        if abs(sec) < 60:
            return str(int(sec)) + " sec"
        elif abs(sec) < (60*60):
            return str(int(sec/60)) + " min " + str(int(sec%60)) + " sec"
        else:
            return str(int(sec/(60*60)))+" hr "+str(int((sec%3600)/60))+" min "+str(int((sec%3600)%60))+" sec"
    def started_time(self):
        return self.start_time
    def elapsed_sec(self):
        return time.time()-self.start_time
    def elapsed_time(self):
        print("Elapsed: %s " % self.elapsed(time.time() - self.start_time))

total_elapsed_time = ElapsedTimer()

import datetime
import os
#os.system("pip install inputimeout nvidia-ml-py3 matplotlib numpy tensorflow-gpu tensorflow-datasets tfds-nightly")

from inputimeout import inputimeout as inpt

date_time_at_start = datetime.datetime.now()

dir_path = os.path.dirname(os.path.realpath(__file__))
print("Your current directory of this python file execution is:",dir_path)

try:
  logging_files = open("Name_of_log_files.txt",'r+')
except:
  logging_files = open("Name_of_log_files.txt",'w')
  logging_files = open("Name_of_log_files.txt",'r+')
list_of_file_names = []
for file_names in logging_files:
      list_of_file_names.append(str(file_names)[:-1])

print("Previously used files:",list_of_file_names)

log_path = os.path.dirname(os.path.realpath(__file__))

same_dir = "Yes"
try:
  same_dir = inpt(prompt="Should the logging file be placed in current dir? (Yes/No):",timeout=15)

  if (same_dir=="No" or same_dir=="no" or same_dir=="N" or same_dir=="n" or same_dir=='0'):
    log_path = inpt(prompt="Directory Name:",timeout=60)
except:
  pass

file_name = log_path + "/Log.txt"
temp = ""

try:
  temp = inpt(prompt="Name for error Logging file(Default is: 'Log.txt'. Leave empty if need to remain default.):",timeout=15)
  if len(temp) >= 1:
   file_name = log_path + "/" + str(temp) + ".txt"
except:
  pass

if file_name not in list_of_file_names:
  logging_files.write(file_name)
  logging_files.write('\n')
  
try:
  # -*- coding: utf-8 -*-
  """Copy of transformer.ipynb

  Automatically generated by Colaboratory.

  Original file is located at
      https://colab.research.google.com/drive/1QAhOYNcBbSSymvEaLzZ385I0ELtO4mER

  ##### Copyright 2019 The TensorFlow Authors.
  """
  
  #@title Licensed under the Apache License, Version 2.0 (the "License");
  # you may not use this file except in compliance with the License.
  # You may obtain a copy of the License at
  #
  # https://www.apache.org/licenses/LICENSE-2.0
  #
  # Unless required by applicable law or agreed to in writing, software
  # distributed under the License is distributed on an "AS IS" BASIS,
  # WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
  # See the License for the specific language governing permissions and
  # limitations under the License.
  
  """# Transformer model for language understanding

  <table class="tfo-notebook-buttons" align="left">
    <td>
      <a target="_blank" href="https://www.tensorflow.org/tutorials/text/transformer">
      <img src="https://www.tensorflow.org/images/tf_logo_32px.png" />
      View on TensorFlow.org</a>
    </td>
    <td>
      <a target="_blank" href="https://colab.research.google.com/github/tensorflow/docs/blob/master/site/en/tutorials/text/transformer.ipynb">
      <img src="https://www.tensorflow.org/images/colab_logo_32px.png" />
      Run in Google Colab</a>
    </td>
    <td>
      <a target="_blank" href="https://github.com/tensorflow/docs/blob/master/site/en/tutorials/text/transformer.ipynb">
      <img src="https://www.tensorflow.org/images/GitHub-Mark-32px.png" />
      View source on GitHub</a>
    </td>
    <td>
      <a href="https://storage.googleapis.com/tensorflow_docs/docs/site/en/tutorials/text/transformer.ipynb"><img src="https://www.tensorflow.org/images/download_logo_32px.png" />Download notebook</a>
    </td>
  </table>

  This tutorial trains a <a href="https://arxiv.org/abs/1706.03762" class="external">Transformer model</a> to translate Portuguese to English. This is an advanced example that assumes knowledge of [text generation](text_generation.ipynb) and [attention](nmt_with_attention.ipynb).

  The core idea behind the Transformer model is *self-attention*—the ability to attend to different positions of the input sequence to compute a representation of that sequence. Transformer creates stacks of self-attention layers and is explained below in the sections *Scaled dot product attention* and *Multi-head attention*.

  A transformer model handles variable-sized input using stacks of self-attention layers instead of [RNNs](text_classification_rnn.ipynb) or [CNNs](../images/intro_to_cnns.ipynb). This general architecture has a number of advantages:

  * It make no assumptions about the temporal/spatial relationships across the data. This is ideal for processing a set of objects (for example, [StarCraft units](https://deepmind.com/blog/alphastar-mastering-real-time-strategy-game-starcraft-ii/#block-8)).
  * Layer outputs can be calculated in parallel, instead of a series like an RNN.
  * Distant items can affect each other's output without passing through many RNN-steps, or convolution layers (see [Scene Memory Transformer](https://arxiv.org/pdf/1903.03878.pdf) for example).
  * It can learn long-range dependencies. This is a challenge in many sequence tasks.

  The downsides of this architecture are:

  * For a time-series, the output for a time-step is calculated from the *entire history* instead of only the inputs and current hidden-state. This _may_ be less efficient.   
  * If the input *does* have a  temporal/spatial relationship, like text, some positional encoding must be added or the model will effectively see a bag of words. 

  After training the model in this notebook, you will be able to input a Portuguese sentence and return the English translation.

  <img src="https://www.tensorflow.org/images/tutorials/transformer/attention_map_portuguese.png" width="800" alt="Attention heatmap">
  """
  
  import tensorflow_datasets as tfds
  import tensorflow as tf

  import numpy as np
  import matplotlib.pyplot as plt

  import nvidia_smi
  nvidia_smi.nvmlInit()
  handle = nvidia_smi.nvmlDeviceGetHandleByIndex(0)
  res = nvidia_smi.nvmlDeviceGetUtilizationRates(handle)
  
  import os,sys
  from tensorflow.python.keras import backend as K
  import gradient_checkpointing.memory_saving_gradients as memory_saving_gradients

  # monkey patch Keras gradients to point to our custom version, with automatic checkpoint selection
  K.__dict__["gradients"] = memory_saving_gradients.gradients_speed

  from tensorflow.python.ops import gradients as tf_gradients
  tf_gradients.gradients = memory_saving_gradients.gradients_speed
  
  tf.config.experimental.set_memory_growth(tf.config.list_physical_devices('GPU')[0],True)

  #tf.config.experimental.set_lms_enabled(True)

  """## Setup input pipeline

  Use [TFDS](https://www.tensorflow.org/datasets) to load the [Portugese-English translation dataset](https://github.com/neulab/word-embeddings-for-nmt) from the [TED Talks Open Translation Project](https://www.ted.com/participate/translate).

  This dataset contains approximately 50000 training examples, 1100 validation examples, and 2000 test examples.
  """
  tffl = tf.float32
  low_precision = False
  low_precision = inpt(prompt="Use low precision float16?(Default is: 'False'. Leave empty if need to remain default.):",timeout=5)
  if low_precision:
    tffl = tf.float16
    policy = tf.keras.mixed_precision.experimental.Policy('mixed_float16')
    tf.keras.mixed_precision.experimental.set_policy(policy)

  examples, metadata = tfds.load('ted_hrlr_translate/pt_to_en', with_info=True,
                                as_supervised=True)
  train_examples, val_examples = examples['train'], examples['validation']
  test_examples = examples['test']
  """Create a custom subwords tokenizer from the training dataset."""
  try:
      tokenizer_pt = tfds.deprecated.text.SubwordTextEncoder.load_from_file("tokenizer_pt")
      tokenizer_en = tfds.deprecated.text.SubwordTextEncoder.load_from_file("tokenizer_en")
  except:
      en = pt = []
      
      for p,e in train_examples:
          pt.append(p.numpy())
          en.append(e.numpy())
      for p,e in val_examples:
          pt.append(p.numpy())
          en.append(e.numpy())
      for p,e in test_examples:
          pt.append(p.numpy())
          en.append(e.numpy())

      tokenizer_en = tfds.deprecated.text.SubwordTextEncoder.build_from_corpus(
          en, target_vocab_size=2**16,max_subword_length=32)

      tokenizer_pt = tfds.deprecated.text.SubwordTextEncoder.build_from_corpus(
          pt, target_vocab_size=2**16,max_subword_length=32)
      tokenizer_pt.save_to_file("tokenizer_pt")
      tokenizer_en.save_to_file("tokenizer_en")

  """The tokenizer encodes the string by breaking it into subwords if the word is not in its dictionary."""

  BUFFER_SIZE = 51785
  BATCH_SIZE = 1

  """Add a start and end token to the input and target."""

  def encode(lang1, lang2):
    lang1 = [tokenizer_pt.vocab_size] + tokenizer_pt.encode(
        lang1.numpy()) + [tokenizer_pt.vocab_size+1]

    lang2 = [tokenizer_en.vocab_size] + tokenizer_en.encode(
        lang2.numpy()) + [tokenizer_en.vocab_size+1]
    
    return lang1, lang2

  """You want to use `Dataset.map` to apply this function to each element of the dataset.  `Dataset.map` runs in graph mode.

  * Graph tensors do not have a value. 
  * In graph mode you can only use TensorFlow Ops and functions. 

  So you can't `.map` this function directly: You need to wrap it in a `tf.py_function`. The `tf.py_function` will pass regular tensors (with a value and a `.numpy()` method to access it), to the wrapped python function.
  """

  def tf_encode(pt, en):
    result_pt, result_en = tf.py_function(encode, [pt, en], [tf.int64, tf.int64])
    result_pt.set_shape([None])
    result_en.set_shape([None])

    return result_pt, result_en

  """Note: To keep this example small and relatively fast, drop examples with a length of over 40 tokens."""

  AVG_LENGTH = 20000
  MAX_LENGTH = 2**16

  def filter_max_length(x, y, max_length=AVG_LENGTH):
    return tf.logical_and(tf.size(x) <= max_length,
                          tf.size(y) <= max_length)
  
  train_dataset = train_examples.map(tf_encode)
  train_dataset = train_dataset.filter(filter_max_length)
  train_dataset = train_dataset.cache("Cache_train")
  train_dataset = train_dataset.shuffle(BUFFER_SIZE).padded_batch(BATCH_SIZE)
  train_dataset = train_dataset.prefetch(tf.data.experimental.AUTOTUNE)


  val_dataset = val_examples.map(tf_encode)
  val_dataset = val_dataset.filter(filter_max_length)
  val_dataset = val_dataset.cache("Cache_val")   
  val_dataset = val_dataset.shuffle(2000).padded_batch(1)
  val_dataset = val_dataset.prefetch(tf.data.experimental.AUTOTUNE)

  test_dataset = test_examples.map(tf_encode)
  test_dataset = test_dataset.filter(filter_max_length)
  test_dataset = test_dataset.cache("Cache_test")   
  test_dataset = test_dataset.shuffle(2000).padded_batch(1)
  test_dataset = test_dataset.prefetch(tf.data.experimental.AUTOTUNE)

  """## Positional encoding

  Since this model doesn't contain any recurrence or convolution, positional encoding is added to give the model some information about the relative position of the words in the sentence. 

  The positional encoding vector is added to the embedding vector. Embeddings represent a token in a d-dimensional space where tokens with similar meaning will be closer to each other. But the embeddings do not encode the relative position of words in a sentence. So after adding the positional encoding, words will be closer to each other based on the *similarity of their meaning and their position in the sentence*, in the d-dimensional space.

  See the notebook on [positional encoding](https://github.com/tensorflow/examples/blob/master/community/en/position_encoding.ipynb) to learn more about it. The formula for calculating the positional encoding is as follows:

  $$\Large{PE_{(pos, 2i)} = sin(pos / 10000^{2i / d_{model}})} $$
  $$\Large{PE_{(pos, 2i+1)} = cos(pos / 10000^{2i / d_{model}})} $$
  """

  def get_angles(pos, i, d_model):
    angle_rates = 1 / np.power(10000, (2 * (i//2)) / np.float16(d_model))
    return pos * angle_rates

  def positional_encoding(position, d_model):
    angle_rads = get_angles(np.arange(position)[:, np.newaxis],
                            np.arange(d_model)[np.newaxis, :],
                            d_model)
    
    # apply sin to even indices in the array; 2i
    angle_rads[:, 0::2] = np.sin(angle_rads[:, 0::2])
    
    # apply cos to odd indices in the array; 2i+1
    angle_rads[:, 1::2] = np.cos(angle_rads[:, 1::2])
      
    pos_encoding = angle_rads[np.newaxis, ...]
      
    return tf.cast(pos_encoding, dtype=tffl)

  """## Masking

  Mask all the pad tokens in the batch of sequence. It ensures that the model does not treat padding as the input. The mask indicates where pad value `0` is present: it outputs a `1` at those locations, and a `0` otherwise.
  """

  def create_padding_mask(seq):
    seq = tf.cast(tf.math.equal(seq, 0), tffl)
    
    # add extra dimensions to add the padding
    # to the attention logits.
    return seq[:, tf.newaxis, tf.newaxis, :]  # (batch_size, 1, 1, seq_len)

  """The look-ahead mask is used to mask the future tokens in a sequence. In other words, the mask indicates which entries should not be used.

  This means that to predict the third word, only the first and second word will be used. Similarly to predict the fourth word, only the first, second and the third word will be used and so on.
  """

  def create_look_ahead_mask(size):
    mask = 1 - tf.linalg.band_part(tf.ones((size, size)), -1, 0)
    return mask  # (seq_len, seq_len)

  """## Scaled dot product attention

  <img src="https://www.tensorflow.org/images/tutorials/transformer/scaled_attention.png" width="500" alt="scaled_dot_product_attention">

  The attention function used by the transformer takes three inputs: Q (query), K (key), V (value). The equation used to calculate the attention weights is:

  $$\Large{Attention(Q, K, V) = softmax_k(\frac{QK^T}{\sqrt{d_k}}) V} $$

  The dot-product attention is scaled by a factor of square root of the depth. This is done because for large values of depth, the dot product grows large in magnitude pushing the softmax function where it has small gradients resulting in a very hard softmax. 

  For example, consider that `Q` and `K` have a mean of 0 and variance of 1. Their matrix multiplication will have a mean of 0 and variance of `dk`. Hence, *square root of `dk`* is used for scaling (and not any other number) because the matmul of `Q` and `K` should have a mean of 0 and variance of 1, and you get a gentler softmax.

  The mask is multiplied with -1e9 (close to negative infinity). This is done because the mask is summed with the scaled matrix multiplication of Q and K and is applied immediately before a softmax. The goal is to zero out these cells, and large negative inputs to softmax are near zero in the output.
  """

  def scaled_dot_product_attention(q, k, v, mask):
    """Calculate the attention weights.
    q, k, v must have matching leading dimensions.
    k, v must have matching penultimate dimension, i.e.: seq_len_k = seq_len_v.
    The mask has different shapes depending on its type(padding or look ahead) 
    but it must be broadcastable for addition.
    
    Args:
      q: query shape == (..., seq_len_q, depth)
      k: key shape == (..., seq_len_k, depth)
      v: value shape == (..., seq_len_v, depth_v)
      mask: Float tensor with shape broadcastable 
            to (..., seq_len_q, seq_len_k). Defaults to None.
      
    Returns:
      output, attention_weights
    """

    matmul_qk = tf.matmul(q, k, transpose_b=True)  # (..., seq_len_q, seq_len_k)
    
    del(q)

    # scale matmul_qk
    dk = tf.cast(tf.shape(k)[-1], tffl)
    scaled_attention_logits = matmul_qk / tf.math.sqrt(dk)

    del(dk,matmul_qk,k)

    # add the mask to the scaled tensor.
    if mask is not None:
      scaled_attention_logits += (mask * -1e9)

    # softmax is normalized on the last axis (seq_len_k) so that the scores
    # add up to 1.
    attention_weights = tf.nn.softmax(scaled_attention_logits, axis=-1)  # (..., seq_len_q, seq_len_k)

    del(scaled_attention_logits)

    output = tf.matmul(attention_weights, v)  # (..., seq_len_q, depth_v)

    del(v)

    return output, attention_weights

  """As the softmax normalization is done on K, its values decide the amount of importance given to Q.

  The output represents the multiplication of the attention weights and the V (value) vector. This ensures that the words you want to focus on are kept as-is and the irrelevant words are flushed out.
  """

  np.set_printoptions(suppress=True)

  """## Multi-head attention

  <img src="https://www.tensorflow.org/images/tutorials/transformer/multi_head_attention.png" width="500" alt="multi-head attention">


  Multi-head attention consists of four parts:
  *    Linear layers and split into heads.
  *    Scaled dot-product attention.
  *    Concatenation of heads.
  *    Final linear layer.

  Each multi-head attention block gets three inputs; Q (query), K (key), V (value). These are put through linear (Dense) layers and split up into multiple heads. 

  The `scaled_dot_product_attention` defined above is applied to each head (broadcasted for efficiency). An appropriate mask must be used in the attention step.  The attention output for each head is then concatenated (using `tf.transpose`, and `tf.reshape`) and put through a final `Dense` layer.

  Instead of one single attention head, Q, K, and V are split into multiple heads because it allows the model to jointly attend to information at different positions from different representational spaces. After the split each head has a reduced dimensionality, so the total computation cost is the same as a single head attention with full dimensionality.
  """

  class MultiHeadAttention(tf.keras.layers.Layer):
    def __init__(self, d_model, num_heads):
      super(MultiHeadAttention, self).__init__()
      self.num_heads = num_heads
      self.d_model = d_model
      
      self.depth = d_model // self.num_heads
      
      self.wq = tf.keras.layers.Dense(d_model)
      self.wk = tf.keras.layers.Dense(d_model)
      self.wv = tf.keras.layers.Dense(d_model)
      
      self.dense = tf.keras.layers.Dense(d_model)
          
    def split_heads(self, x, batch_size):
      """Split the last dimension into (num_heads, depth).
      Transpose the result such that the shape is (batch_size, num_heads, seq_len, depth)
      """
      x = tf.reshape(x, (batch_size, -1, self.num_heads, self.depth))
      return tf.transpose(x, perm=[0, 2, 1, 3])
      
    def call(self, v, k, q, mask):
      batch_size = tf.shape(q)[0]
      
      q = self.wq(q)  # (batch_size, seq_len, d_model)
      k = self.wk(k)  # (batch_size, seq_len, d_model)
      v = self.wv(v)  # (batch_size, seq_len, d_model)
      
      q = self.split_heads(q, batch_size)  # (batch_size, num_heads, seq_len_q, depth)
      k = self.split_heads(k, batch_size)  # (batch_size, num_heads, seq_len_k, depth)
      v = self.split_heads(v, batch_size)  # (batch_size, num_heads, seq_len_v, depth)
      
      # scaled_attention.shape == (batch_size, num_heads, seq_len_q, depth)
      # attention_weights.shape == (batch_size, num_heads, seq_len_q, seq_len_k)
      scaled_attention, attention_weights = scaled_dot_product_attention(
          q, k, v, mask)
      
      del(q,k,v)

      scaled_attention = tf.transpose(scaled_attention, perm=[0, 2, 1, 3])  # (batch_size, seq_len_q, num_heads, depth)

      concat_attention = tf.reshape(scaled_attention, 
                                    (batch_size, -1, self.d_model))  # (batch_size, seq_len_q, d_model)

      del(scaled_attention)

      output = self.dense(concat_attention)  # (batch_size, seq_len_q, d_model)
          
      del(concat_attention)

      return output, attention_weights

  """## Point wise feed forward network

  Point wise feed forward network consists of two fully-connected layers with a ReLU activation in between.
  """

  def point_wise_feed_forward_network(d_model, dff):
    return tf.keras.Sequential([
        tf.keras.layers.Dense(dff, activation='relu'),  # (batch_size, seq_len, dff)
        tf.keras.layers.Dense(d_model)  # (batch_size, seq_len, d_model)
    ])

  """## Encoder and decoder

  <img src="https://www.tensorflow.org/images/tutorials/transformer/transformer.png" width="600" alt="transformer">

  The transformer model follows the same general pattern as a standard [sequence to sequence with attention model](nmt_with_attention.ipynb). 

  * The input sentence is passed through `N` encoder layers that generates an output for each word/token in the sequence.
  * The decoder attends on the encoder's output and its own input (self-attention) to predict the next word.

  ### Encoder layer

  Each encoder layer consists of sublayers:

  1.   Multi-head attention (with padding mask) 
  2.    Point wise feed forward networks. 

  Each of these sublayers has a residual connection around it followed by a layer normalization. Residual connections help in avoiding the vanishing gradient problem in deep networks.

  The output of each sublayer is `LayerNorm(x + Sublayer(x))`. The normalization is done on the `d_model` (last) axis. There are N encoder layers in the transformer.
  """

  class EncoderLayer(tf.keras.layers.Layer):
    def __init__(self, d_model, num_heads, dff, rate=0.1):
      super(EncoderLayer, self).__init__()

      self.mha = MultiHeadAttention(d_model, num_heads)
      self.ffn = point_wise_feed_forward_network(d_model, dff)

      self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)
      self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)
      
      self.dropout1 = tf.keras.layers.Dropout(rate)
      self.dropout2 = tf.keras.layers.Dropout(rate)
      
    def call(self, x, training, mask):

      attn_output, _ = self.mha(x, x, x, mask)  # (batch_size, input_seq_len, d_model)
      attn_output = self.dropout1(attn_output, training=training)
      out1 = self.layernorm1(x + attn_output)  # (batch_size, input_seq_len, d_model)
      
      ffn_output = self.ffn(out1)  # (batch_size, input_seq_len, d_model)
      ffn_output = self.dropout2(ffn_output, training=training)
      out2 = self.layernorm2(out1 + ffn_output)  # (batch_size, input_seq_len, d_model)
      
      return out2

  """### Decoder layer

  Each decoder layer consists of sublayers:

  1.   Masked multi-head attention (with look ahead mask and padding mask)
  2.   Multi-head attention (with padding mask). V (value) and K (key) receive the *encoder output* as inputs. Q (query) receives the *output from the masked multi-head attention sublayer.*
  3.   Point wise feed forward networks

  Each of these sublayers has a residual connection around it followed by a layer normalization. The output of each sublayer is `LayerNorm(x + Sublayer(x))`. The normalization is done on the `d_model` (last) axis.

  There are N decoder layers in the transformer.

  As Q receives the output from decoder's first attention block, and K receives the encoder output, the attention weights represent the importance given to the decoder's input based on the encoder's output. In other words, the decoder predicts the next word by looking at the encoder output and self-attending to its own output. See the demonstration above in the scaled dot product attention section.
  """

  class DecoderLayer(tf.keras.layers.Layer):
    def __init__(self, d_model, num_heads, dff, rate=0.1):
      super(DecoderLayer, self).__init__()

      self.mha1 = MultiHeadAttention(d_model, num_heads)
      self.mha2 = MultiHeadAttention(d_model, num_heads)

      self.ffn = point_wise_feed_forward_network(d_model, dff)
  
      self.layernorm1 = tf.keras.layers.LayerNormalization(epsilon=1e-6)
      self.layernorm2 = tf.keras.layers.LayerNormalization(epsilon=1e-6)
      self.layernorm3 = tf.keras.layers.LayerNormalization(epsilon=1e-6)
      
      self.dropout1 = tf.keras.layers.Dropout(rate)
      self.dropout2 = tf.keras.layers.Dropout(rate)
      self.dropout3 = tf.keras.layers.Dropout(rate)
      
      
    def call(self, x, enc_output, training, 
            look_ahead_mask, padding_mask):
      # enc_output.shape == (batch_size, input_seq_len, d_model)

      attn1, attn_weights_block1 = self.mha1(x, x, x, look_ahead_mask)  # (batch_size, target_seq_len, d_model)
      attn1 = self.dropout1(attn1, training=training)
      out1 = self.layernorm1(attn1 + x)
      
      attn2, attn_weights_block2 = self.mha2(
          enc_output, enc_output, out1, padding_mask)  # (batch_size, target_seq_len, d_model)
      attn2 = self.dropout2(attn2, training=training)
      out2 = self.layernorm2(attn2 + out1)  # (batch_size, target_seq_len, d_model)
      
      ffn_output = self.ffn(out2)  # (batch_size, target_seq_len, d_model)
      ffn_output = self.dropout3(ffn_output, training=training)
      out3 = self.layernorm3(ffn_output + out2)  # (batch_size, target_seq_len, d_model)
      
      return out3, attn_weights_block1, attn_weights_block2

  """### Encoder

  The `Encoder` consists of:
  1.   Input Embedding
  2.   Positional Encoding
  3.   N encoder layers

  The input is put through an embedding which is summed with the positional encoding. The output of this summation is the input to the encoder layers. The output of the encoder is the input to the decoder.
  """

  class Encoder(tf.keras.layers.Layer):
    def __init__(self, num_layers, d_model, num_heads, dff, input_vocab_size,
                maximum_position_encoding, rate=0.1):
      super(Encoder, self).__init__()

      self.d_model = d_model
      self.num_layers = num_layers
      
      self.embedding = tf.keras.layers.Embedding(input_vocab_size, d_model)
      self.pos_encoding = positional_encoding(maximum_position_encoding, 
                                              self.d_model)
      
      
      self.enc_layers = [EncoderLayer(d_model, num_heads, dff, rate) 
                        for _ in range(num_layers)]
      #self.enc_layers = [tf.recompute_grad(f) for f in self.enc_layers_]
    
      self.dropout = tf.keras.layers.Dropout(rate)
          
    def call(self, x, training, mask):

      seq_len = tf.shape(x)[1]
      
      # adding embedding and position encoding.
      x = self.embedding(x)  # (batch_size, input_seq_len, d_model)
      x *= tf.math.sqrt(tf.cast(self.d_model, tffl))
      x += self.pos_encoding[:, :seq_len, :]

      x = self.dropout(x, training=training)
      
      for i in range(self.num_layers):
        x = self.enc_layers[i](x, training, mask)
      
      return x  # (batch_size, input_seq_len, d_model)

  """### Decoder

  The `Decoder` consists of:
  1.   Output Embedding
  2.   Positional Encoding
  3.   N decoder layers

  The target is put through an embedding which is summed with the positional encoding. The output of this summation is the input to the decoder layers. The output of the decoder is the input to the final linear layer.
  """

  class Decoder(tf.keras.layers.Layer):
    def __init__(self, num_layers, d_model, num_heads, dff, target_vocab_size,
                maximum_position_encoding, rate=0.1):
      super(Decoder, self).__init__()

      self.d_model = d_model
      self.num_layers = num_layers
      
      self.embedding = tf.keras.layers.Embedding(target_vocab_size, d_model)
      self.pos_encoding = positional_encoding(maximum_position_encoding, d_model)
      
      self.dec_layers = [DecoderLayer(d_model, num_heads, dff, rate) 
                        for _ in range(num_layers)]
      #self.dec_layers = [tf.recompute_grad(f) for f in self.dec_layers_]
      
      self.dropout = tf.keras.layers.Dropout(rate)
      
    def call(self, x, enc_output, training, 
            look_ahead_mask, padding_mask):

      seq_len = tf.shape(x)[1]
      attention_weights = {}
      
      x = self.embedding(x)  # (batch_size, target_seq_len, d_model)
      x *= tf.math.sqrt(tf.cast(self.d_model, tffl))
      x += self.pos_encoding[:, :seq_len, :]

      x = self.dropout(x, training=training)

      for i in range(self.num_layers):
        x, block1, block2 = self.dec_layers[i](x, enc_output, training,
                                              look_ahead_mask, padding_mask)
        
        attention_weights['decoder_layer{}_block1'.format(i+1)] = block1
        attention_weights['decoder_layer{}_block2'.format(i+1)] = block2
      
      # x.shape == (batch_size, target_seq_len, d_model)
      return x, attention_weights

  """## Create the Transformer

  Transformer consists of the encoder, decoder and a final linear layer. The output of the decoder is the input to the linear layer and its output is returned.
  """

  class Transformer(tf.keras.Model):
    def __init__(self, num_layers, d_model, num_heads, dff, input_vocab_size, 
                target_vocab_size, pe_input, pe_target, rate=0.1):
      super(Transformer, self).__init__()

      self.num_heads = num_heads

      self.encoder = Encoder(num_layers, d_model, num_heads, dff, 
                            input_vocab_size, pe_input, rate)

      self.decoder = Decoder(num_layers, d_model, num_heads, dff, 
                            target_vocab_size, pe_target, rate)

      self.final_layer = tf.keras.layers.Dense(target_vocab_size)
      
    def call(self, inp, tar, training, enc_padding_mask, 
            look_ahead_mask, dec_padding_mask):

      enc_output = self.encoder(inp, training, enc_padding_mask)  # (batch_size, inp_seq_len, d_model)
      
      # dec_output.shape == (batch_size, tar_seq_len, d_model)
      dec_output, attention_weights = self.decoder(
          tar, enc_output, training, look_ahead_mask, dec_padding_mask)

      final_output = tf.cast(self.final_layer(dec_output), tf.float32)  # (batch_size, tar_seq_len, target_vocab_size)
      
      return final_output, attention_weights

  """## Set hyperparameters

  To keep this example small and relatively fast, the values for *num_layers, d_model, and dff* have been reduced. 

  The values used in the base model of transformer were; *num_layers=6*, *d_model = 512*, *dff = 2048*. See the [paper](https://arxiv.org/abs/1706.03762) for all the other versions of the transformer.

  Note: By changing the values below, you can get the model that achieved state of the art on many tasks.
  """

  num_layers = 64
  d_model = dff = 512
  num_heads = 32

  input_vocab_size =  max(tokenizer_pt.vocab_size + 2,tokenizer_en.vocab_size + 2)
  target_vocab_size = max(tokenizer_pt.vocab_size + 2,tokenizer_en.vocab_size + 2)
  dropout_rate = 0.3

  """## Optimizer

  Use the Adam optimizer with a custom learning rate scheduler according to the formula in the [paper](https://arxiv.org/abs/1706.03762).

  $$\Large{lrate = d_{model}^{-0.5} * min(step{\_}num^{-0.5}, step{\_}num * warmup{\_}steps^{-1.5})}$$
  """
  current_time = datetime.datetime.now().strftime("%Y%m%d-%H%M%S")
  logg = 'logdir/' + current_time + "/learning rate"
  lr_summary_writer = tf.summary.create_file_writer(logg)

  class CustomSchedule(tf.keras.optimizers.schedules.LearningRateSchedule):
    def __init__(self, d_model, warmup_steps=4000):
      super(CustomSchedule, self).__init__()
      
      self.d_model = d_model
      self.d_model = tf.cast(self.d_model, tffl)

      self.warmup_steps = warmup_steps
      
    def __call__(self, step):
      arg1 = tf.math.rsqrt(step)
      arg2 = step * (self.warmup_steps ** -1.5)
      
      learning_rate = tf.cast(tf.math.rsqrt(self.d_model), tffl) * tf.cast(tf.math.minimum(arg1, arg2), tffl)
      learning_rate = tf.cast(learning_rate, tffl)

      with lr_summary_writer.as_default():
        tf.summary.scalar('learning rate', data=learning_rate, step=tf.cast(step,tf.int64))
      
      return learning_rate

  learning_rate = CustomSchedule(d_model, warmup_steps=100000//BATCH_SIZE)

  optimizer = tf.keras.optimizers.Adam(learning_rate, beta_1=0.9, beta_2=0.98, 
                                      epsilon=1e-9)

  """## Loss and metrics

  Since the target sequences are padded, it is important to apply a padding mask when calculating the loss.
  """

  loss_object = tf.keras.losses.SparseCategoricalCrossentropy(
      from_logits=True, reduction='none')

  def loss_function(real, pred):
    mask = tf.math.logical_not(tf.math.equal(real, 0))
    loss_ = loss_object(real, pred)

    mask = tf.cast(mask, dtype=loss_.dtype)
    loss_ *= mask
    
    return tf.reduce_sum(loss_)/tf.reduce_sum(mask)

  train_loss = tf.keras.metrics.Mean(name='train_loss')
  train_accuracy = tf.keras.metrics.SparseCategoricalAccuracy(
      name='train_accuracy')

  val_loss = tf.keras.metrics.Mean(name='val_loss')
  val_accuracy = tf.keras.metrics.SparseCategoricalAccuracy(
      name='val_accuracy')

  test_loss = tf.keras.metrics.Mean(name='test_loss')
  test_accuracy = tf.keras.metrics.SparseCategoricalAccuracy(
      name='test_accuracy')

  """## Training and checkpointing"""

  transformer = Transformer(num_layers, d_model, num_heads, dff,
                            input_vocab_size, target_vocab_size, 
                            pe_input=MAX_LENGTH, 
                            pe_target=MAX_LENGTH,
                            rate=dropout_rate)

  def create_masks(inp, tar):
    # Encoder padding mask
    enc_padding_mask = create_padding_mask(inp)
    
    # Used in the 2nd attention block in the decoder.
    # This padding mask is used to mask the encoder outputs.
    dec_padding_mask = create_padding_mask(inp)
    
    # Used in the 1st attention block in the decoder.
    # It is used to pad and mask future tokens in the input received by 
    # the decoder.
    look_ahead_mask = tf.cast(create_look_ahead_mask(tf.shape(tar)[1]),tffl)
    dec_target_padding_mask = create_padding_mask(tar)
    combined_mask = tf.maximum(dec_target_padding_mask, look_ahead_mask)
    
    return enc_padding_mask, combined_mask, dec_padding_mask

  """Create the checkpoint path and the checkpoint manager. This will be used to save checkpoints every `n` epochs."""

  checkpoint_path = "./checkpoints/train/model" + "_" +  str(num_layers) + "_" + str(d_model) + "_" + str(dff) + "_" + str(num_heads) + "_" + str(dropout_rate)
  model_save_path = checkpoint_path + "/Saved_Model_Complete"

  ckpt = tf.train.Checkpoint(transformer=transformer,
                            optimizer=optimizer)

  ckpt_manager = tf.train.CheckpointManager(ckpt, checkpoint_path, max_to_keep=5)

  # if a checkpoint exists, restore the latest checkpoint.
  if ckpt_manager.latest_checkpoint:
    try:
      transformer = tf.keras.models.load_model(model_save_path)
    except:
      ckpt.restore(ckpt_manager.latest_checkpoint)
      print ('Latest checkpoint restored!!')

  """The target is divided into tar_inp and tar_real. tar_inp is passed as an input to the decoder. `tar_real` is that same input shifted by 1: At each location in `tar_input`, `tar_real` contains the  next token that should be predicted.

  For example, `sentence` = "SOS A lion in the jungle is sleeping EOS"

  `tar_inp` =  "SOS A lion in the jungle is sleeping"

  `tar_real` = "A lion in the jungle is sleeping EOS"

  The transformer is an auto-regressive model: it makes predictions one part at a time, and uses its output so far to decide what to do next. 

  During training this example uses teacher-forcing (like in the [text generation tutorial](./text_generation.ipynb)). Teacher forcing is passing the true output to the next time step regardless of what the model predicts at the current time step.

  As the transformer predicts each word, *self-attention* allows it to look at the previous words in the input sequence to better predict the next word.

  To prevent the model from peeking at the expected output the model uses a look-ahead mask.
  """

  EPOCHS = 0

  # The @tf.function trace-compiles train_step into a TF graph for faster
  # execution. The function specializes to the precise shape of the argument
  # tensors. To avoid re-tracing due to the variable sequence lengths or variable
  # batch sizes (the last batch is smaller), use input_signature to specify
  # more generic shapes.

  train_step_signature = [
      tf.TensorSpec(shape=(None, None), dtype=tf.int64),
      tf.TensorSpec(shape=(None, None), dtype=tf.int64),
  ]

  @tf.function(input_signature=train_step_signature,experimental_relax_shapes=True)
  def train_step(inp, tar):
    tar_inp = tar[:, :-1]
    tar_real = tar[:, 1:]
    
    enc_padding_mask, combined_mask, dec_padding_mask = create_masks(inp, tar_inp)
    
    with tf.GradientTape() as tape:
      predictions, _ = transformer(inp, tar_inp, 
                                  True, 
                                  enc_padding_mask, 
                                  combined_mask, 
                                  dec_padding_mask)
      loss = loss_function(tar_real, predictions)

    gradients = tape.gradient(loss, transformer.trainable_variables)    
    optimizer.apply_gradients(zip(gradients, transformer.trainable_variables))
    
    train_loss(loss)
    train_accuracy(tar_real, predictions)

  val_step_signature = [
      tf.TensorSpec(shape=(None, None), dtype=tf.int64),
      tf.TensorSpec(shape=(None, None), dtype=tf.int64),
  ]
  @tf.function(input_signature=val_step_signature,experimental_relax_shapes=True)
  def val_step(inp, tar):
    tar_inp = tar[:, :-1]
    tar_real = tar[:, 1:]

    print("Validating...",sep="",end="",flush=True)
  
    for i in "Validating....":
      print("\b",flush=True,end="")
    
    enc_padding_mask, combined_mask, dec_padding_mask = create_masks(inp, tar_inp)
    
    predictions, _ = transformer(inp, tar_inp, 
                                False, 
                                enc_padding_mask, 
                                combined_mask, 
                                dec_padding_mask)
    loss = loss_function(tar_real, predictions)

    val_loss(loss)
    val_accuracy(tar_real, predictions)

  test_step_signature = [
      tf.TensorSpec(shape=(None, None), dtype=tf.int64),
      tf.TensorSpec(shape=(None, None), dtype=tf.int64),
  ]
  @tf.function(input_signature=test_step_signature,experimental_relax_shapes=True)
  def test_step(inp, tar):
    tar_inp = tar[:, :-1]
    tar_real = tar[:, 1:]
    
    print("Testing...",sep="",end="",flush=True)
  
    for i in "Testing....":
      print("\b",flush=True,end="")

    enc_padding_mask, combined_mask, dec_padding_mask = create_masks(inp, tar_inp)
    
    predictions, _ = transformer(inp, tar_inp, 
                                False, 
                                enc_padding_mask, 
                                combined_mask, 
                                dec_padding_mask)
    loss = loss_function(tar_real, predictions)

    test_loss(loss)
    test_accuracy(tar_real, predictions)
  
  train_log_dir = 'logdir/' + current_time + '/train'
  val_log_dir = 'logdir/' + current_time + '/val'
  test_log_dir = 'logdir/' + current_time + '/test'
  logg = 'logdir/' + current_time + "/"
  train_summary_writer = tf.summary.create_file_writer(train_log_dir)
  test_summary_writer = tf.summary.create_file_writer(test_log_dir)
  val_summary_writer = tf.summary.create_file_writer(val_log_dir)

  """Portuguese is used as the input language and English is the target language."""

  test_loss.reset_states()
  test_accuracy.reset_states()
    
  ETA_sum = 0

  step_num = 1
  epoch = 0
  tf.summary.trace_on(graph=True, profiler=True)
  #tf.profiler.experimental.start(logg)
  while True:
    
    epoch_time = ElapsedTimer()
    
    train_loss.reset_states()
    train_accuracy.reset_states()
      
    val_loss.reset_states()
    val_accuracy.reset_states()

    # inp -> portuguese, tar -> english
    for (batch, (inp, tar)) in enumerate(train_dataset):
      batch_time = ElapsedTimer()
      #if batch == 10:
        #tf.profiler.experimental.stop()

      #with tf.profiler.experimental.Trace('Train',step_num=optimizer.iterations.numpy(), _r =1):
      train_step(inp, tar)
      #train_step(tar, inp)

      step_num += 1

      with train_summary_writer.as_default():
        tf.summary.scalar('loss', train_loss.result(), step=optimizer.iterations.numpy())
        tf.summary.scalar('accuracy', train_accuracy.result(), step=optimizer.iterations.numpy())

      if not epoch and not batch :
            transformer.summary()
      
      if ((batch+1)%((51785/BATCH_SIZE)//100)==0 or batch==0):
        print ('Epoch {} Batch {}. Training-Loss {:.4f} Training-Accuracy {:.4f}. Percent of epoch done {}\t'.format(
            epoch + 1, batch, train_loss.result(), train_accuracy.result(),((batch+1)//((51785/BATCH_SIZE)//100))))
      """
      if ((batch+1)%((51785/BATCH_SIZE)//10)==0 and not (batch+1)//((51785/BATCH_SIZE)//10)==10 or batch == 51785//BATCH_SIZE):
        ckpt_save_path = ckpt_manager.save()
        print ('Saving checkpoint for epoch {} at {}\t'.format(epoch+1,
                                                          ckpt_save_path))
      """
      ETA = (batch_time.elapsed_sec() * (51785/BATCH_SIZE)) // 1
      if (batch > 2) :
        ETA_sum += ETA
        ETA_avg = ( ETA_sum / (batch-2) ) - epoch_time.elapsed_sec()
    
        if (abs(ETA_avg) < 60):
          ETA_avg = "Average ETA to this epoch completion: "+str(int(ETA_avg)) + " sec"
        elif (abs(ETA_avg) < (60*60)):
          ETA_avg = "Average ETA to this epoch completion: "+str(int(ETA_avg/60)) + " min " + str(int(ETA_avg%60)) + " sec"
        else:
          ETA_avg = "Average ETA to this epoch completion: "+str(int(ETA_avg/(60*60)))+" hr "+str(int((ETA_avg%3600)/60))+" min "+str(int((ETA_avg%3600)%60))+" sec"
    
        print(ETA_avg,sep="",end="",flush=True)
    
        for i in range(len(ETA_avg)+1):
          print("\b",flush=True,end="")
    
    ckpt_save_path = ckpt_manager.save()
    print ('Saving checkpoint for epoch {} at {}\t'.format(epoch+1,
                                                          ckpt_save_path))

    ETA_sum /= 51783//BATCH_SIZE
    for (batch,(i,t)) in enumerate(val_dataset):
        val_step(i,t)
        #val_step(t, i)
    with val_summary_writer.as_default():
      tf.summary.scalar('loss', val_loss.result(), step=optimizer.iterations.numpy())
      tf.summary.scalar('accuracy', val_accuracy.result(), step=optimizer.iterations.numpy())

    print ('Epoch {} Validation-Loss {:.4f} Validation-Accuracy {:.4f}\t'.format(
            epoch + 1, val_loss.result(), val_accuracy.result()))
      
    print ('Epoch {} Training-Loss {:.4f} Training-Accuracy {:.4f}\t'.format(epoch + 1, 
                                                  train_loss.result(), 
                                                  train_accuracy.result()))

    print ('Time taken for this epoch: {}\t'.format(epoch_time.elapsed(epoch_time.elapsed_sec())))

    epoch += 1

    if (float(train_accuracy.result()) >= 0.4 or float(val_accuracy.result()) > 0.869):
      print("Train Accuracy or Validation accuracy is very-high, stopping the training...")
      with train_summary_writer.as_default():
        tf.summary.trace_export(name="train_trace",step=optimizer.iterations.numpy(),profiler_outdir=train_log_dir)
      break

  for (batch,(inp_test,tar_test)) in enumerate(test_dataset):
      test_step(inp_test,tar_test)
      #test_step(tar_test, inp_test)
  with test_summary_writer.as_default():
    tf.summary.scalar('loss', test_loss.result(), step=optimizer.iterations.numpy())
    tf.summary.scalar('accuracy', test_accuracy.result(), step=optimizer.iterations.numpy())
  
  
  print('Test-Loss {:.4f} Test-Accuracy {:.4f}'.format(
          test_loss.result(), test_accuracy.result()))
  try:
    if (float(test_accuracy.result()) > 0.8 ):
          transformer.save(model_save_path)
    else:
          print("Accuray quite low, to give a meaningful result.\n Test Accuracy:",float(test_accuracy.result()))
  except Exception as e:
    try:
      f = open(file_name,'r+')
    except:
      f = open(file_name,'w')
      f = open(file_name,'r+')
    last_count = None
    try:
      for line in f:
        last_count = line
      last_count = int(last_count)
    except:
      last_count = 0
    last_count += 1
    f.write(str(date_time_at_start))
    f.write("{")
    f.write(str(e))
    f.write("}\n")
    f.write(str(last_count))
    f.write("\n")
    print("Total no. of errors done since last reset:",last_count)
    print("To check the errors, see the file:",file_name)

  """## Evaluate

  The following steps are used for evaluation:

  * Encode the input sentence using the Portuguese tokenizer (`tokenizer_pt`). Moreover, add the start and end token so the input is equivalent to what the model is trained with. This is the encoder input.
  * The decoder input is the `start token == tokenizer_en.vocab_size`.
  * Calculate the padding masks and the look ahead masks.
  * The `decoder` then outputs the predictions by looking at the `encoder output` and its own output (self-attention).
  * Select the last word and calculate the argmax of that.
  * Concatentate the predicted word to the decoder input as pass it to the decoder.
  * In this approach, the decoder predicts the next word based on the previous words it predicted.

  Note: The model used here has less capacity to keep the example relatively faster so the predictions maybe less right. To reproduce the results in the paper, use the entire dataset and base transformer model or transformer XL, by changing the hyperparameters above.
  """

  def evaluate(inp_sentence):
    start_token = [tokenizer_pt.vocab_size]
    end_token = [tokenizer_pt.vocab_size + 1]
    
    memory_usage = []

    # inp sentence is portuguese, hence adding the start and end token
    inp_sentence = start_token + tokenizer_pt.encode(inp_sentence) + end_token
    encoder_input = tf.expand_dims(inp_sentence, 0)
    
    # as the target is english, the first word to the transformer should be the
    # english start token.
    decoder_input = [tokenizer_en.vocab_size]
    output = tf.expand_dims(decoder_input, 0)
      
    for i in range(MAX_LENGTH):
      enc_padding_mask, combined_mask, dec_padding_mask = create_masks(
          encoder_input, output)
      memory_usage.append(res.memory)
      
      # predictions.shape == (batch_size, seq_len, vocab_size)
      predictions, attention_weights = transformer(encoder_input, 
                                                  output,
                                                  False,
                                                  enc_padding_mask,
                                                  combined_mask,
                                                  dec_padding_mask)
      
      # select the last word from the seq_len dimension
      predictions = predictions[: ,-1:, :]  # (batch_size, 1, vocab_size)

      predicted_id = tf.cast(tf.argmax(predictions, axis=-1), tf.int32)
      
      # return the result if the predicted_id is equal to the end token
      if predicted_id == tokenizer_en.vocab_size+1:
        return tf.squeeze(output, axis=0), attention_weights, memory_usage
      
      # concatentate the predicted_id to the output which is given to the decoder
      # as its input.
      output = tf.concat([output, predicted_id], axis=-1)

    return tf.squeeze(output, axis=0), attention_weights, memory_usage

  def plot_attention_weights(attention, sentence, result, layer):
    fig = plt.figure(figsize=(16, 8))
    
    sentence = tokenizer_pt.encode(sentence)
    
    attention = tf.squeeze(attention[layer], axis=0)
    
    for head in range(attention.shape[0]):
      ax = fig.add_subplot(2, 4, head+1)
      
      # plot the attention weights
      ax.matshow(attention[head][:-1, :], cmap='viridis')

      fontdict = {'fontsize': 10}
      
      ax.set_xticks(range(len(sentence)+2))
      ax.set_yticks(range(len(result)))
      
      ax.set_ylim(len(result)-1.5, -0.5)
          
      ax.set_xticklabels(
          ['<start>']+[tokenizer_pt.decode([i]) for i in sentence]+['<end>'], 
          fontdict=fontdict, rotation=90)
      
      ax.set_yticklabels([tokenizer_en.decode([i]) for i in result 
                          if i < tokenizer_en.vocab_size], 
                        fontdict=fontdict)
      
      ax.set_xlabel('Head {}'.format(head+1))
    
    plt.tight_layout()
    plt.show()

  def translate(sentence, plot=''):
    result, attention_weights, memory_usage = evaluate(sentence)
    
    predicted_sentence = tokenizer_en.decode([i for i in result 
                                              if i < tokenizer_en.vocab_size])  

    print('Input: {}'.format(sentence))
    print('Predicted translation: {}'.format(predicted_sentence))
    #print("Memory Usage with increasing length", memory_usage)
    
    if plot:
      plot_attention_weights(attention_weights, sentence, result, plot)

  sen = ""
  for i in "este é um problema que temos que resolver.":
        sen += i
        print(sen)
        translate(sen)
        inp = input(":::")

  translate("este é um problema que temos que resolver.")
  print ("Real translation: this is a problem we have to solve .")

  translate("os meus vizinhos ouviram sobre esta ideia.")
  print ("Real translation: and my neighboring homes heard about this idea .")

  translate("vou então muito rapidamente partilhar convosco algumas histórias de algumas coisas mágicas que aconteceram.")
  print ("Real translation: so i 'll just share with you some stories very quickly of some magical things that have happened .")

  """You can pass different layers and attention blocks of the decoder to the `plot` parameter."""

  translate("este é o primeiro livro que eu fiz.")
  print ("Real translation: this is the first book i've ever done.")
except Exception as e:
  try:
    f = open(file_name,'r+')
  except:
    f = open(file_name,'w')
    f = open(file_name,'r+')
  last_count = None
  try:
    for line in f:
      last_count = line
    last_count = int(last_count)
  except:
    last_count = 0
  last_count += 1
  f.write(str(date_time_at_start))
  f.write("{")
  f.write(str(e))
  f.write("}\n")
  f.write(str(last_count))
  f.write("\n")
  print("Total no. of errors done since last reset:",last_count)
  print("To check the errors, see the file:",file_name)
finally:
  total_elapsed_time.elapsed_time()