# Natural Language Processing

In [3]:
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import tensorflow as tf

from google.colab import drive

drive.mount('/drive')

Mounted at /drive


## Part 1 - Data

In [4]:
path_to_file = "/drive/My Drive/Colab Notebooks/Deep Learning Course (Jose Portilia)/7. Natural Language Processing/shakespeare.txt"

In [5]:
text = open(path_to_file, mode = 'r').read()

In [6]:
text[:500]

"\n                     1\n  From fairest creatures we desire increase,\n  That thereby beauty's rose might never die,\n  But as the riper should by time decease,\n  His tender heir might bear his memory:\n  But thou contracted to thine own bright eyes,\n  Feed'st thy light's flame with self-substantial fuel,\n  Making a famine where abundance lies,\n  Thy self thy foe, to thy sweet self too cruel:\n  Thou that art now the world's fresh ornament,\n  And only herald to the gaudy spring,\n  Within thine own bu"

In [7]:
print(text[:500])


                     1
  From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But as the riper should by time decease,
  His tender heir might bear his memory:
  But thou contracted to thine own bright eyes,
  Feed'st thy light's flame with self-substantial fuel,
  Making a famine where abundance lies,
  Thy self thy foe, to thy sweet self too cruel:
  Thou that art now the world's fresh ornament,
  And only herald to the gaudy spring,
  Within thine own bu


In [8]:
sorted(set(text))[:10]

['\n', ' ', '!', '"', '&', "'", '(', ')', ',', '-']

In [9]:
vocab = sorted(set(text))

In [10]:
len(vocab)

84

## Part 2 - Text Preprocessing

In [11]:
char_to_ind = {char:i for i, char in enumerate(vocab)}

In [12]:
char_to_ind['H']

33

In [13]:
ind_to_char = np.array(vocab)

In [14]:
ind_to_char[33]

'H'

In [15]:
encoded_text = np.array([char_to_ind[c] for c in text])

In [16]:
encoded_text

array([ 0,  1,  1, ..., 30, 39, 29])

In [17]:
encoded_text.shape

(5445609,)

In [18]:
len(text)

5445609

In [19]:
text[:100]

"\n                     1\n  From fairest creatures we desire increase,\n  That thereby beauty's rose mi"

In [20]:
encoded_text[:100]

array([ 0,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,
        1,  1,  1,  1,  1, 12,  0,  1,  1, 31, 73, 70, 68,  1, 61, 56, 64,
       73, 60, 74, 75,  1, 58, 73, 60, 56, 75, 76, 73, 60, 74,  1, 78, 60,
        1, 59, 60, 74, 64, 73, 60,  1, 64, 69, 58, 73, 60, 56, 74, 60,  8,
        0,  1,  1, 45, 63, 56, 75,  1, 75, 63, 60, 73, 60, 57, 80,  1, 57,
       60, 56, 76, 75, 80,  5, 74,  1, 73, 70, 74, 60,  1, 68, 64])

## Part 3 - Creating Batches

In [21]:
print(text[:500])


                     1
  From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But as the riper should by time decease,
  His tender heir might bear his memory:
  But thou contracted to thine own bright eyes,
  Feed'st thy light's flame with self-substantial fuel,
  Making a famine where abundance lies,
  Thy self thy foe, to thy sweet self too cruel:
  Thou that art now the world's fresh ornament,
  And only herald to the gaudy spring,
  Within thine own bu


In [22]:
line = "From fairest creatures we desire increase"

In [23]:
len(line)

41

In [24]:
lines = """
From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But as the riper should by time decease,
"""

In [25]:
len(lines)

133

In [26]:
seq_len = 120  ## to capture 3 lines

In [27]:
total_num_seq = len(text) // (seq_len + 1)

In [28]:
total_num_seq

45005

In [29]:
char_dataset = tf.data.Dataset.from_tensor_slices(encoded_text)

In [30]:
type(char_dataset)

tensorflow.python.data.ops.dataset_ops.TensorSliceDataset

In [31]:
for item in char_dataset.take(500):
  print(ind_to_char[item.numpy()])



 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1


 
 
F
r
o
m
 
f
a
i
r
e
s
t
 
c
r
e
a
t
u
r
e
s
 
w
e
 
d
e
s
i
r
e
 
i
n
c
r
e
a
s
e
,


 
 
T
h
a
t
 
t
h
e
r
e
b
y
 
b
e
a
u
t
y
'
s
 
r
o
s
e
 
m
i
g
h
t
 
n
e
v
e
r
 
d
i
e
,


 
 
B
u
t
 
a
s
 
t
h
e
 
r
i
p
e
r
 
s
h
o
u
l
d
 
b
y
 
t
i
m
e
 
d
e
c
e
a
s
e
,


 
 
H
i
s
 
t
e
n
d
e
r
 
h
e
i
r
 
m
i
g
h
t
 
b
e
a
r
 
h
i
s
 
m
e
m
o
r
y
:


 
 
B
u
t
 
t
h
o
u
 
c
o
n
t
r
a
c
t
e
d
 
t
o
 
t
h
i
n
e
 
o
w
n
 
b
r
i
g
h
t
 
e
y
e
s
,


 
 
F
e
e
d
'
s
t
 
t
h
y
 
l
i
g
h
t
'
s
 
f
l
a
m
e
 
w
i
t
h
 
s
e
l
f
-
s
u
b
s
t
a
n
t
i
a
l
 
f
u
e
l
,


 
 
M
a
k
i
n
g
 
a
 
f
a
m
i
n
e
 
w
h
e
r
e
 
a
b
u
n
d
a
n
c
e
 
l
i
e
s
,


 
 
T
h
y
 
s
e
l
f
 
t
h
y
 
f
o
e
,
 
t
o
 
t
h
y
 
s
w
e
e
t
 
s
e
l
f
 
t
o
o
 
c
r
u
e
l
:


 
 
T
h
o
u
 
t
h
a
t
 
a
r
t
 
n
o
w
 
t
h
e
 
w
o
r
l
d
'
s
 
f
r
e
s
h
 
o
r
n
a
m
e
n
t
,


 
 
A
n
d
 
o
n
l
y
 
h
e
r
a
l
d
 
t
o
 
t
h
e
 
g
a
u
d
y
 
s
p
r
i
n
g
,


 
 
W
i
t
h
i
n
 
t
h
i
n
e
 
o
w
n
 
b
u


In [32]:
sequences = char_dataset.batch(seq_len + 1, drop_remainder = True)

In [33]:
def create_seq_targets(seq):    ## Hello
 
  input_text = seq[:-1]     ## Hell
  target_text = seq[1:]     ## ello

  return input_text, target_text

In [34]:
dataset = sequences.map(create_seq_targets)

In [35]:
for input_text, target_text in dataset.take(1):
  print(input_text.numpy())
  print("".join(ind_to_char[input_text.numpy()]))
  print()
  print(target_text.numpy())
  print("".join(ind_to_char[target_text.numpy()]))

[ 0  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1 12  0
  1  1 31 73 70 68  1 61 56 64 73 60 74 75  1 58 73 60 56 75 76 73 60 74
  1 78 60  1 59 60 74 64 73 60  1 64 69 58 73 60 56 74 60  8  0  1  1 45
 63 56 75  1 75 63 60 73 60 57 80  1 57 60 56 76 75 80  5 74  1 73 70 74
 60  1 68 64 62 63 75  1 69 60 77 60 73  1 59 64 60  8  0  1  1 27 76 75]

                     1
  From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But

[ 1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1 12  0  1
  1 31 73 70 68  1 61 56 64 73 60 74 75  1 58 73 60 56 75 76 73 60 74  1
 78 60  1 59 60 74 64 73 60  1 64 69 58 73 60 56 74 60  8  0  1  1 45 63
 56 75  1 75 63 60 73 60 57 80  1 57 60 56 76 75 80  5 74  1 73 70 74 60
  1 68 64 62 63 75  1 69 60 77 60 73  1 59 64 60  8  0  1  1 27 76 75  1]
                     1
  From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But 


In [36]:
batch_size = 128

In [37]:
buffer_size = 10000

dataset = dataset.shuffle(buffer_size = buffer_size).batch(batch_size, drop_remainder = True)

In [38]:
dataset

<BatchDataset shapes: ((128, 120), (128, 120)), types: (tf.int64, tf.int64)>

## Part 4 - Creating a Model

### Some Variables

**Will need them later**

In [39]:
vocab_size = len(vocab)

In [40]:
vocab_size

84

In [41]:
embedd_dim = 64

In [42]:
rnn_neurons = 1026

### Specifying Loss Function

In [43]:
from tensorflow.keras.losses import sparse_categorical_crossentropy

In [44]:
help(sparse_categorical_crossentropy)

Help on function sparse_categorical_crossentropy in module keras.losses:

sparse_categorical_crossentropy(y_true, y_pred, from_logits=False, axis=-1)
    Computes the sparse categorical crossentropy loss.
    
    Standalone usage:
    
    >>> y_true = [1, 2]
    >>> y_pred = [[0.05, 0.95, 0], [0.1, 0.8, 0.1]]
    >>> loss = tf.keras.losses.sparse_categorical_crossentropy(y_true, y_pred)
    >>> assert loss.shape == (2,)
    >>> loss.numpy()
    array([0.0513, 2.303], dtype=float32)
    
    Args:
      y_true: Ground truth values.
      y_pred: The predicted values.
      from_logits: Whether `y_pred` is expected to be a logits tensor. By default,
        we assume that `y_pred` encodes a probability distribution.
      axis: Defaults to -1. The dimension along which the entropy is
        computed.
    
    Returns:
      Sparse categorical crossentropy loss value.



In [45]:
def sparse_cat_loss(y_true, y_pred):
  return sparse_categorical_crossentropy(y_true, y_pred, from_logits = True)

### Creating Model

In [46]:
from tensorflow.keras.models import Sequential
from tensorflow.keras.layers import Embedding, GRU, Dense

In [47]:
def create_model(vocab_size, embedd_dim, rnn_neurons, batch_size):
  model = Sequential()

  model.add(Embedding(input_dim = vocab_size, output_dim = embedd_dim, batch_input_shape = [batch_size, None]))

  model.add(GRU(units = rnn_neurons, return_sequences = True, stateful = True, recurrent_initializer = 'glorot_uniform'))

  model.add(Dense(units = vocab_size))

  model.compile(loss = sparse_cat_loss, optimizer = 'adam')

  return model

In [48]:
model = create_model(vocab_size = vocab_size, embedd_dim = embedd_dim, rnn_neurons = rnn_neurons, batch_size = batch_size)

In [49]:
model.summary()

Model: "sequential"
_________________________________________________________________
 Layer (type)                Output Shape              Param #   
 embedding (Embedding)       (128, None, 64)           5376      
                                                                 
 gru (GRU)                   (128, None, 1026)         3361176   
                                                                 
 dense (Dense)               (128, None, 84)           86268     
                                                                 
Total params: 3,452,820
Trainable params: 3,452,820
Non-trainable params: 0
_________________________________________________________________


## Part 5 - Training the Model

### Checking if Model is working

**By predicting without training it should output some random text

In [50]:
for input_example_batch, target_example_batch in dataset.take(1):

  example_batch_pred = model(input_example_batch)

  print(example_batch_pred.shape, " <=== (batch_size, sequence_length, vocab_size)")

(128, 120, 84)  <=== (batch_size, sequence_length, vocab_size)


In [51]:
sampled_ind = tf.random.categorical(logits = example_batch_pred[0], num_samples = 1)

In [52]:
sampled_ind = tf.squeeze(sampled_ind, axis = -1).numpy()

In [53]:
sampled_ind

array([77, 80, 35, 35, 25, 80, 38, 58, 37, 22, 17, 29, 24, 57, 18, 57, 36,
       31, 47, 54, 16, 61,  1, 75, 31, 80, 76, 74, 72, 27, 23, 45, 32, 11,
       42,  8, 15, 79, 11, 58, 59, 38, 14,  1, 74, 49, 63, 59, 63, 23, 14,
       13, 23, 53, 45, 38, 79, 52,  7, 16, 36,  8, 39, 65, 73, 42, 59, 24,
       69, 60,  3, 19, 13, 51, 71, 81, 63, 58, 53, 83, 70,  9, 18, 11, 14,
       60, 74, 62, 34, 58, 39, 30, 25, 79,  6, 32, 43, 37, 72, 66, 64, 14,
        2, 43, 23, 39, 26, 82, 76, 18, 68, 50, 66, 45, 49, 21, 73,  2, 72,
       55])

In [54]:
"".join(ind_to_char[sampled_ind])

'vyJJ?yMcL;6D>b7bKFV_5f tFyusqB<TG0Q,4x0cdM3 sXhdh<32<]TMx[)5K,NjrQd>ne"82Zpzhc]}o-703esgIcNE?x(GRLqki3!R<NA|u7mYkTX:r!q`'

### Training Model

In [55]:
epochs = 30

In [56]:
model.fit(dataset, epochs = epochs)

Epoch 1/30
Epoch 2/30
Epoch 3/30
Epoch 4/30
Epoch 5/30
Epoch 6/30
Epoch 7/30
Epoch 8/30
Epoch 9/30
Epoch 10/30
Epoch 11/30
Epoch 12/30
Epoch 13/30
Epoch 14/30
Epoch 15/30
Epoch 16/30
Epoch 17/30
Epoch 18/30
Epoch 19/30
Epoch 20/30
Epoch 21/30
Epoch 22/30
Epoch 23/30
Epoch 24/30
Epoch 25/30
Epoch 26/30
Epoch 27/30
Epoch 28/30
Epoch 29/30
Epoch 30/30


<keras.callbacks.History at 0x7f9145b993d0>

In [65]:
model.save("/drive/My Drive/Colab Notebooks/Deep Learning Course (Jose Portilia)/7. Natural Language Processing/NLP_model.h5")

## Part 6 - Text Generation

In [66]:
new_model = create_model(vocab_size, embedd_dim, rnn_neurons, batch_size = 1)

new_model.load_weights('/drive/My Drive/Colab Notebooks/Deep Learning Course (Jose Portilia)/7. Natural Language Processing/NLP_model.h5')


new_model.build(tf.TensorShape([1, None]))

In [67]:
new_model.summary()

Model: "sequential_2"
_________________________________________________________________
 Layer (type)                Output Shape              Param #   
 embedding_2 (Embedding)     (1, None, 64)             5376      
                                                                 
 gru_2 (GRU)                 (1, None, 1026)           3361176   
                                                                 
 dense_2 (Dense)             (1, None, 84)             86268     
                                                                 
Total params: 3,452,820
Trainable params: 3,452,820
Non-trainable params: 0
_________________________________________________________________


In [68]:
def generate_text(model, start_seed, gen_size = 500, temp = 1.0):

  num_generate = gen_size

  input_eval = [char_to_ind[s] for s in start_seed]

  input_eval = tf.expand_dims(input_eval, 0)

  text_generated = []

  temperature = temp

  model.reset_states()

  for i in range(num_generate):

    predictions = model(input_eval)

    predictions =  tf.squeeze(predictions, 0)

    predictions = predictions/temperature

    predicted_id = tf.random.categorical(predictions, num_samples = 1)[-1, 0].numpy()

    input_eval = tf.expand_dims([predicted_id], 0)

    text_generated.append(ind_to_char[predicted_id])

  return (start_seed + "".join(text_generated))

In [69]:
print(generate_text(new_model, "JULIET", gen_size = 1000))

JULIET. Nay, then she impediment!
    I know what-seeding is sharp in a speech,
    Now boughes blue loud derigs? Hang his unnocent
    At Troy, nay, and briefly now guess'd their sons,
    Where you survive enof the wren of good condition lodely in my moving there! Do
    Seal up our drums before him?
  GLOUCESTER. Thou hast enough were set; I must go see a chamber;
    And so approve it, sir, would wish me from your safety, fly her. Mean
    of my heart, that you can hold question to your eyes of yours
    Though that shape is sland'red in brazorous from a mine.
    Come, come, but from three hundred yellow into the world; and if
    I do not find it out; he spake of nothing of yourself
    As bravely should be joyful times, unswelp'd his grief
    Still dear tricking, as it were a between babe
    Unfix my brother of his hearing is of mine.
  FALSTAFF. Farewell!                  Exit
  CLOWN. Alas, it is not one of our master; they are outron
    To make for suitors for moves. Thou 