# Text Generation with Neural Networks

In [3]:
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt

import tensorflow as tf

## Data

Website : https://www.gutenberg.org/

All of shakespeare's works

A very large corpus of text (Large number of text date required for realistic text generation).

In [4]:
text = open('/content/shakespeare.txt', mode='r').read()

In [5]:
text[:500]

"\n                     1\n  From fairest creatures we desire increase,\n  That thereby beauty's rose might never die,\n  But as the riper should by time decease,\n  His tender heir might bear his memory:\n  But thou contracted to thine own bright eyes,\n  Feed'st thy light's flame with self-substantial fuel,\n  Making a famine where abundance lies,\n  Thy self thy foe, to thy sweet self too cruel:\n  Thou that art now the world's fresh ornament,\n  And only herald to the gaudy spring,\n  Within thine own bu"

## Unique characters

In [6]:
vocab = sorted(set(text))
print(vocab)

['\n', ' ', '!', '"', '&', "'", '(', ')', ',', '-', '.', '0', '1', '2', '3', '4', '5', '6', '7', '8', '9', ':', ';', '<', '>', '?', 'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M', 'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z', '[', ']', '_', '`', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z', '|']


In [7]:

len(vocab)

83

## Text Processing

### Indexing 

In [8]:
char_to_index = {char:index for index,char in enumerate(vocab)}

In [9]:
char_to_index

{'\n': 0,
 ' ': 1,
 '!': 2,
 '"': 3,
 '&': 4,
 "'": 5,
 '(': 6,
 ')': 7,
 ',': 8,
 '-': 9,
 '.': 10,
 '0': 11,
 '1': 12,
 '2': 13,
 '3': 14,
 '4': 15,
 '5': 16,
 '6': 17,
 '7': 18,
 '8': 19,
 '9': 20,
 ':': 21,
 ';': 22,
 '<': 23,
 '>': 24,
 '?': 25,
 'A': 26,
 'B': 27,
 'C': 28,
 'D': 29,
 'E': 30,
 'F': 31,
 'G': 32,
 'H': 33,
 'I': 34,
 'J': 35,
 'K': 36,
 'L': 37,
 'M': 38,
 'N': 39,
 'O': 40,
 'P': 41,
 'Q': 42,
 'R': 43,
 'S': 44,
 'T': 45,
 'U': 46,
 'V': 47,
 'W': 48,
 'X': 49,
 'Y': 50,
 'Z': 51,
 '[': 52,
 ']': 53,
 '_': 54,
 '`': 55,
 'a': 56,
 'b': 57,
 'c': 58,
 'd': 59,
 'e': 60,
 'f': 61,
 'g': 62,
 'h': 63,
 'i': 64,
 'j': 65,
 'k': 66,
 'l': 67,
 'm': 68,
 'n': 69,
 'o': 70,
 'p': 71,
 'q': 72,
 'r': 73,
 's': 74,
 't': 75,
 'u': 76,
 'v': 77,
 'w': 78,
 'x': 79,
 'y': 80,
 'z': 81,
 '|': 82}

In [10]:
char_to_index['1']

12

In [11]:
index_to_char = np.array(vocab)

In [12]:
index_to_char

array(['\n', ' ', '!', '"', '&', "'", '(', ')', ',', '-', '.', '0', '1',
       '2', '3', '4', '5', '6', '7', '8', '9', ':', ';', '<', '>', '?',
       'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
       'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
       '[', ']', '_', '`', 'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i',
       'j', 'k', 'l', 'm', 'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v',
       'w', 'x', 'y', 'z', '|'], dtype='<U1')

In [13]:
index_to_char[44]

'S'

## Encoding

In [14]:
encoded_text = np.array([char_to_index[c] for c in text])

In [15]:
encoded_text

array([ 0,  1,  1, ..., 78, 63, 64])

In [16]:
encoded_text.shape

(2097152,)

In [17]:
sample = text[:500]
sample

"\n                     1\n  From fairest creatures we desire increase,\n  That thereby beauty's rose might never die,\n  But as the riper should by time decease,\n  His tender heir might bear his memory:\n  But thou contracted to thine own bright eyes,\n  Feed'st thy light's flame with self-substantial fuel,\n  Making a famine where abundance lies,\n  Thy self thy foe, to thy sweet self too cruel:\n  Thou that art now the world's fresh ornament,\n  And only herald to the gaudy spring,\n  Within thine own bu"

In [18]:
encoded_sample = encoded_text[:500]
encoded_sample

array([ 0,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,  1,
        1,  1,  1,  1,  1, 12,  0,  1,  1, 31, 73, 70, 68,  1, 61, 56, 64,
       73, 60, 74, 75,  1, 58, 73, 60, 56, 75, 76, 73, 60, 74,  1, 78, 60,
        1, 59, 60, 74, 64, 73, 60,  1, 64, 69, 58, 73, 60, 56, 74, 60,  8,
        0,  1,  1, 45, 63, 56, 75,  1, 75, 63, 60, 73, 60, 57, 80,  1, 57,
       60, 56, 76, 75, 80,  5, 74,  1, 73, 70, 74, 60,  1, 68, 64, 62, 63,
       75,  1, 69, 60, 77, 60, 73,  1, 59, 64, 60,  8,  0,  1,  1, 27, 76,
       75,  1, 56, 74,  1, 75, 63, 60,  1, 73, 64, 71, 60, 73,  1, 74, 63,
       70, 76, 67, 59,  1, 57, 80,  1, 75, 64, 68, 60,  1, 59, 60, 58, 60,
       56, 74, 60,  8,  0,  1,  1, 33, 64, 74,  1, 75, 60, 69, 59, 60, 73,
        1, 63, 60, 64, 73,  1, 68, 64, 62, 63, 75,  1, 57, 60, 56, 73,  1,
       63, 64, 74,  1, 68, 60, 68, 70, 73, 80, 21,  0,  1,  1, 27, 76, 75,
        1, 75, 63, 70, 76,  1, 58, 70, 69, 75, 73, 56, 58, 75, 60, 59,  1,
       75, 70,  1, 75, 63

## Creating Batches

model will try to predict the next highest probability character given a historical sequence of characters. 


Length of historic sequence


    Too short a sequence and don't have enough information (e.g. given the letter "a" , what is the next character)
    too long a sequence and training will take too long and most likely overfit to sequence characters that are irrelevant to characters farther out. 
+ While there is no correct sequence length choice, consider 
    + the text itself, 
    + how long normal phrases are in it, and 
    + a reasonable idea of what characters/words are relevant to each other.

## Understanding the text 

In [19]:
print(text[:500])


                     1
  From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But as the riper should by time decease,
  His tender heir might bear his memory:
  But thou contracted to thine own bright eyes,
  Feed'st thy light's flame with self-substantial fuel,
  Making a famine where abundance lies,
  Thy self thy foe, to thy sweet self too cruel:
  Thou that art now the world's fresh ornament,
  And only herald to the gaudy spring,
  Within thine own bu


In [20]:
lines = 'From fairest creatures we desire increase,'
len(lines)

42

As Shakespears writing's include rhymes between the lines.
Pick up at least 3 lines to pick up those patterns.

Example: check out the ending words. They do rhyme, such as `increase, decrease` , `die , eyes, lies`

In [21]:
lines = '''
 From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But as the riper should by time decease,
'''

len(lines)

134

## Training Sequences

The actual text data will be the text sequence shifted one character forward. For example:

**Sequence In: "Hello my nam"**

**Sequence Out: "ello my name"**


We can use the `tf.data.Dataset.from_tensor_slices` function to convert a text vector into a stream of character indices.

In [22]:
sequence_length = 120

In [23]:
total_num_sequence = len(text) // (sequence_length+1)

In [24]:
total_num_sequence

17331

In [25]:
# Create Training Sequences
char_dataset = tf.data.Dataset.from_tensor_slices(encoded_text)

In [26]:
type(char_dataset)

tensorflow.python.data.ops.dataset_ops.TensorSliceDataset

In [27]:
for item in char_dataset.take(500):
    print(index_to_char[item.numpy()])



 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1


 
 
F
r
o
m
 
f
a
i
r
e
s
t
 
c
r
e
a
t
u
r
e
s
 
w
e
 
d
e
s
i
r
e
 
i
n
c
r
e
a
s
e
,


 
 
T
h
a
t
 
t
h
e
r
e
b
y
 
b
e
a
u
t
y
'
s
 
r
o
s
e
 
m
i
g
h
t
 
n
e
v
e
r
 
d
i
e
,


 
 
B
u
t
 
a
s
 
t
h
e
 
r
i
p
e
r
 
s
h
o
u
l
d
 
b
y
 
t
i
m
e
 
d
e
c
e
a
s
e
,


 
 
H
i
s
 
t
e
n
d
e
r
 
h
e
i
r
 
m
i
g
h
t
 
b
e
a
r
 
h
i
s
 
m
e
m
o
r
y
:


 
 
B
u
t
 
t
h
o
u
 
c
o
n
t
r
a
c
t
e
d
 
t
o
 
t
h
i
n
e
 
o
w
n
 
b
r
i
g
h
t
 
e
y
e
s
,


 
 
F
e
e
d
'
s
t
 
t
h
y
 
l
i
g
h
t
'
s
 
f
l
a
m
e
 
w
i
t
h
 
s
e
l
f
-
s
u
b
s
t
a
n
t
i
a
l
 
f
u
e
l
,


 
 
M
a
k
i
n
g
 
a
 
f
a
m
i
n
e
 
w
h
e
r
e
 
a
b
u
n
d
a
n
c
e
 
l
i
e
s
,


 
 
T
h
y
 
s
e
l
f
 
t
h
y
 
f
o
e
,
 
t
o
 
t
h
y
 
s
w
e
e
t
 
s
e
l
f
 
t
o
o
 
c
r
u
e
l
:


 
 
T
h
o
u
 
t
h
a
t
 
a
r
t
 
n
o
w
 
t
h
e
 
w
o
r
l
d
'
s
 
f
r
e
s
h
 
o
r
n
a
m
e
n
t
,


 
 
A
n
d
 
o
n
l
y
 
h
e
r
a
l
d
 
t
o
 
t
h
e
 
g
a
u
d
y
 
s
p
r
i
n
g
,


 
 
W
i
t
h
i
n
 
t
h
i
n
e
 
o
w
n
 
b
u


### Creating Sequence Batches

 `Drop_remainder` :

    Drop_remainder: (Optional.) A `tf.bool` scalar `tf.Tensor`, representing
    whether the last batch should be dropped in the case it has fewer than
    `batch_size` elements; the default behavior is not to drop the smaller
    batch.

In [28]:
sequences = char_dataset.batch(sequence_length+1, drop_remainder=True)

In [29]:
sequences

<BatchDataset element_spec=TensorSpec(shape=(121,), dtype=tf.int64, name=None)>

####Create target text sequences:

Grab the input text sequence

Assign the target text sequence as the input text sequence shifted by one step forward

Group them together as a tuple

In [30]:
def create_sequence_targets(seq):
    input_txt = seq[:-1] # Hello my nam
    target_txt = seq[1:] # ello my name
    return (input_txt, target_txt)

In [31]:
dataset = sequences.map(tf.autograph.experimental.do_not_convert(create_sequence_targets))

In [32]:
for input_txt, target_txt in dataset.take(1):

    print(input_txt.numpy())
    print("".join(index_to_char[input_txt.numpy()]))
    
    print('\n')
    
    
    print(target_txt.numpy())
    print("".join(index_to_char[target_txt.numpy()]))

[ 0  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1 12  0
  1  1 31 73 70 68  1 61 56 64 73 60 74 75  1 58 73 60 56 75 76 73 60 74
  1 78 60  1 59 60 74 64 73 60  1 64 69 58 73 60 56 74 60  8  0  1  1 45
 63 56 75  1 75 63 60 73 60 57 80  1 57 60 56 76 75 80  5 74  1 73 70 74
 60  1 68 64 62 63 75  1 69 60 77 60 73  1 59 64 60  8  0  1  1 27 76 75]

                     1
  From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But


[ 1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1  1 12  0  1
  1 31 73 70 68  1 61 56 64 73 60 74 75  1 58 73 60 56 75 76 73 60 74  1
 78 60  1 59 60 74 64 73 60  1 64 69 58 73 60 56 74 60  8  0  1  1 45 63
 56 75  1 75 63 60 73 60 57 80  1 57 60 56 76 75 80  5 74  1 73 70 74 60
  1 68 64 62 63 75  1 69 60 77 60 73  1 59 64 60  8  0  1  1 27 76 75  1]
                     1
  From fairest creatures we desire increase,
  That thereby beauty's rose might never die,
  But 


### Generating Training Batches

In [33]:
batch_size = 128

buffer_size = 1000

dataset = dataset.shuffle(buffer_size).batch(batch_size, drop_remainder=True)


In [34]:
dataset

<BatchDataset element_spec=(TensorSpec(shape=(128, 120), dtype=tf.int64, name=None), TensorSpec(shape=(128, 120), dtype=tf.int64, name=None))>

## Creating the Model

In [35]:
# Length of the vocabulary in chars
vocab_size = len(vocab)

# The embedding dimension 
embed_dim = 64

# Number of RNN units
rnn_neurons = 1026

In [36]:
vocab_size 

83

In [37]:
from tensorflow.keras.losses import sparse_categorical_crossentropy

In [38]:
def sparse_cat_loss(y_true, y_pred):
    return sparse_categorical_crossentropy(y_true, y_pred, from_logits=True)

In [39]:
from tensorflow.keras.models import Sequential
from tensorflow.keras.layers import Dense, LSTM, Embedding, Dropout, GRU

In [40]:
#Setting up Model Function

def create_model(vocab_size, embed_dim, rnn_neurons, batch_size):
    model = Sequential()
    
    model.add(Embedding(vocab_size, embed_dim, batch_input_shape=[batch_size, None]))
    
    model.add(GRU(rnn_neurons,
                          return_sequences=True,
                          stateful=True,
                          recurrent_initializer='glorot_uniform'))
    
    model.add(Dense(vocab_size))
    
    model.compile(optimizer='adam', loss=sparse_cat_loss)
    
    return model

In [41]:
model = create_model(vocab_size, embed_dim, rnn_neurons, batch_size)

In [42]:
model.summary()

Model: "sequential"
_________________________________________________________________
 Layer (type)                Output Shape              Param #   
 embedding (Embedding)       (128, None, 64)           5312      
                                                                 
 gru (GRU)                   (128, None, 1026)         3361176   
                                                                 
 dense (Dense)               (128, None, 83)           85241     
                                                                 
Total params: 3,451,729
Trainable params: 3,451,729
Non-trainable params: 0
_________________________________________________________________


## Training the model


In [43]:
for input_example_batch, target_example_batch in dataset.take(1):

  # Predict off some random batch
  example_batch_predictions = model(input_example_batch)

  # Display the dimensions of the predictions
  print(example_batch_predictions.shape, " <=== (batch_size, sequence_length, vocab_size)")

(128, 120, 83)  <=== (batch_size, sequence_length, vocab_size)


In [44]:
example_batch_predictions.shape

TensorShape([128, 120, 83])

In [45]:
example_batch_predictions[0]

<tf.Tensor: shape=(120, 83), dtype=float32, numpy=
array([[ 4.2124642e-03,  7.3780920e-03,  4.3778191e-04, ...,
        -2.8053438e-03, -1.5491958e-03, -2.2309874e-03],
       [-2.0576089e-04,  3.1742894e-03,  7.0315355e-04, ...,
         2.2278122e-04, -3.1104316e-03,  1.0511335e-03],
       [-1.8069755e-03,  5.4872746e-04,  1.7769056e-04, ...,
         2.0268445e-03, -4.6364199e-03,  2.1188839e-03],
       ...,
       [ 4.9558384e-03,  5.3273477e-03, -4.6511053e-04, ...,
        -2.0526445e-03, -5.7539325e-03, -2.6582757e-03],
       [ 1.5547297e-04,  1.6597235e-03, -9.7347991e-05, ...,
         6.7029591e-04, -6.0118446e-03,  7.4897712e-04],
       [-1.7472229e-03, -5.4388103e-04, -3.3778269e-04, ...,
         2.2391872e-03, -6.4067361e-03,  1.9473135e-03]], dtype=float32)>

Probabilities for each characters


In [46]:
sampled_indices = tf.random.categorical(example_batch_predictions[0], num_samples=1)

In [47]:
sampled_indices

<tf.Tensor: shape=(120, 1), dtype=int64, numpy=
array([[55],
       [50],
       [79],
       [24],
       [10],
       [31],
       [30],
       [46],
       [50],
       [38],
       [53],
       [71],
       [10],
       [61],
       [33],
       [22],
       [16],
       [37],
       [72],
       [33],
       [61],
       [39],
       [14],
       [51],
       [15],
       [78],
       [27],
       [31],
       [62],
       [72],
       [71],
       [20],
       [61],
       [ 9],
       [ 9],
       [66],
       [22],
       [57],
       [73],
       [ 7],
       [37],
       [19],
       [ 4],
       [67],
       [29],
       [13],
       [16],
       [ 6],
       [12],
       [23],
       [69],
       [75],
       [37],
       [29],
       [63],
       [61],
       [60],
       [46],
       [25],
       [49],
       [ 1],
       [38],
       [24],
       [ 7],
       [38],
       [47],
       [14],
       [ 4],
       [72],
       [40],
       [58],
       [45],
       [26],
   

In [48]:
# Reformat to not be a lists of lists
sampled_indices = tf.squeeze(sampled_indices,axis=-1).numpy()

In [49]:
sampled_indices

array([55, 50, 79, 24, 10, 31, 30, 46, 50, 38, 53, 71, 10, 61, 33, 22, 16,
       37, 72, 33, 61, 39, 14, 51, 15, 78, 27, 31, 62, 72, 71, 20, 61,  9,
        9, 66, 22, 57, 73,  7, 37, 19,  4, 67, 29, 13, 16,  6, 12, 23, 69,
       75, 37, 29, 63, 61, 60, 46, 25, 49,  1, 38, 24,  7, 38, 47, 14,  4,
       72, 40, 58, 45, 26, 16, 49, 32, 37, 10, 62, 32,  3, 53, 76, 64,  1,
       48, 81,  2, 28, 40, 71,  7, 38, 36, 13, 38,  3, 57, 80, 45, 20, 17,
       30, 75, 55, 14, 36, 13, 49, 72, 72, 68, 38, 48, 17, 60, 18, 30, 42,
       67])

In [50]:
print("Given the input seq: \n")
print("".join(index_to_char[input_example_batch[0]]))
print('\n')
print("Next Char Predictions: \n")
print("".join(index_to_char[sampled_indices ]))


Given the input seq: 


    Shall profit thee, and much enrich thy book.


                     78
  So oft have I invoked thee for my muse,
  


Next Char Predictions: 

`Yx>.FEUYM]p.fH;5LqHfN3Z4wBFgqp9f--k;br)L8&lD25(1<ntLDhfeU?X M>)MV3&qOcTA5XGL.gG"]ui Wz!COp)MK2M"byT96Et`3K2XqqmMW6e7EQl


These are just random predictions of characters as model hasn't trained.



In [51]:
epochs = 30

In [52]:
model.fit(dataset, epochs = epochs)

Epoch 1/30
Epoch 2/30
Epoch 3/30
Epoch 4/30
Epoch 5/30
Epoch 6/30
Epoch 7/30
Epoch 8/30
Epoch 9/30
Epoch 10/30
Epoch 11/30
Epoch 12/30
Epoch 13/30
Epoch 14/30
Epoch 15/30
Epoch 16/30
Epoch 17/30
Epoch 18/30
Epoch 19/30
Epoch 20/30
Epoch 21/30
Epoch 22/30
Epoch 23/30
Epoch 24/30
Epoch 25/30
Epoch 26/30
Epoch 27/30
Epoch 28/30
Epoch 29/30
Epoch 30/30


<keras.callbacks.History at 0x7f68fa530e50>

## Generating text

The model only expects 128 sequences at a time. 

Creating a model that only expects a batch_size=1.

Then call .build() on the model

In [53]:
model.save('/content/shakespeare_gen.h5')

In [54]:
from tensorflow.keras.models import load_model

In [55]:
model = create_model(vocab_size, embed_dim, rnn_neurons, batch_size=1)

model.load_weights('/content/shakespeare_gen.h5')

model.build(tf.TensorShape([1, None]))

In [56]:
model.summary()

Model: "sequential_1"
_________________________________________________________________
 Layer (type)                Output Shape              Param #   
 embedding_1 (Embedding)     (1, None, 64)             5312      
                                                                 
 gru_1 (GRU)                 (1, None, 1026)           3361176   
                                                                 
 dense_1 (Dense)             (1, None, 83)             85241     
                                                                 
Total params: 3,451,729
Trainable params: 3,451,729
Non-trainable params: 0
_________________________________________________________________


In [57]:
def generate_text(model, start_seed,gen_size=100,temp=1.0):
  '''
  model: Trained Model to Generate Text
  start_seed: Intial Seed text in string form
  gen_size: Number of characters to generate

  Basic idea behind this function is to take in some seed text, format it so
  that it is in the correct shape for our network, then loop the sequence as
  we keep adding our own predicted characters. Similar to our work in the RNN
  time series problems.
  '''

  # Number of characters to generate
  num_generate = gen_size

  # Vecotrizing starting seed text
  input_eval = [char_to_index[s] for s in start_seed]

  # Expand to match batch format shape
  input_eval = tf.expand_dims(input_eval, 0)

  # Empty list to hold resulting generated text
  text_generated = []

  # Temperature effects randomness in our resulting text
  # The term is derived from entropy/thermodynamics.
  # The temperature is used to effect probability of next characters.
  # Higher probability == lesss surprising/ more expected
  # Lower temperature == more surprising / less expected
 
  temperature = temp

  # Here batch size == 1
  model.reset_states()

  for i in range(num_generate):

      # Generate Predictions
      predictions = model(input_eval)

      # Remove the batch shape dimension
      predictions = tf.squeeze(predictions, 0)

      # Use a cateogircal disitribution to select the next character
      predictions = predictions / temperature
      predicted_id = tf.random.categorical(predictions, num_samples=1)[-1,0].numpy()

      # Pass the predicted charracter for the next input
      input_eval = tf.expand_dims([predicted_id], 0)

      # Transform back to character letter
      text_generated.append(index_to_char[predicted_id])

  return (start_seed + ''.join(text_generated))

In [61]:
print(generate_text(model,"love",gen_size=100))

love it you?
  EDWARD. A wife to bring your looks, and life; but when a lowly rack,
    My wife and lobl
