In [1]:
# Naive LSTM to learn three-char window to one-char mapping
import numpy
from keras.models import Sequential
from keras.layers import Dense
from keras.layers import LSTM
from keras.utils import np_utils

Using TensorFlow backend.


In [2]:
# fix random seed for reproducibility
numpy.random.seed(7)

In [3]:
# define the raw dataset
alphabet = "ABCDEFGHIJKLMNOPQRSTUVWXYZ"
# create mapping of characters to integers (0-25) and the reverse
char_to_int = dict((c, i) for i, c in enumerate(alphabet))
int_to_char = dict((i, c) for i, c in enumerate(alphabet))
# prepare the dataset of input to output pairs encoded as integers
seq_length = 3
dataX = []
dataY = []
for i in range(0, len(alphabet) - seq_length, 1):
    seq_in = alphabet[i:i + seq_length]
    seq_out = alphabet[i + seq_length]
    dataX.append([char_to_int[char] for char in seq_in])
    dataY.append(char_to_int[seq_out])
    print(seq_in, '->', seq_out)

ABC -> D
BCD -> E
CDE -> F
DEF -> G
EFG -> H
FGH -> I
GHI -> J
HIJ -> K
IJK -> L
JKL -> M
KLM -> N
LMN -> O
MNO -> P
NOP -> Q
OPQ -> R
PQR -> S
QRS -> T
RST -> U
STU -> V
TUV -> W
UVW -> X
VWX -> Y
WXY -> Z


In [4]:
# reshape X to be [samples, time steps, features]
X = numpy.reshape(dataX, (len(dataX), seq_length,1))
# normalize
X = X / float(len(alphabet))
# one hot encode the output variable
y = np_utils.to_categorical(dataY)

In [6]:
# create and fit the model
model = Sequential()
model.add(LSTM(32, input_shape=(X.shape[1], X.shape[2])))
model.add(Dense(y.shape[1], activation='softmax'))
model.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])
model.fit(X, y, epochs=500, batch_size=1, verbose=2)
# summarize performance of the model
scores = model.evaluate(X, y, verbose=0)
print("Model Accuracy: %.2f%%" % (scores[1]*100))
# demonstrate some model predictions
for pattern in dataX:
    x = numpy.reshape(pattern, (1, len(pattern), 1))
    x = x / float(len(alphabet))
    prediction = model.predict(x, verbose=0)
    index = numpy.argmax(prediction)
    result = int_to_char[index]
    seq_in = [int_to_char[value] for value in pattern]
    print(seq_in, "->", result)

Epoch 1/500
0s - loss: 3.2775 - acc: 0.0435
Epoch 2/500
0s - loss: 3.2622 - acc: 0.0435
Epoch 3/500
0s - loss: 3.2536 - acc: 0.0435
Epoch 4/500
0s - loss: 3.2454 - acc: 0.0435
Epoch 5/500
0s - loss: 3.2370 - acc: 0.0435
Epoch 6/500
0s - loss: 3.2296 - acc: 0.0435
Epoch 7/500
0s - loss: 3.2197 - acc: 0.0435
Epoch 8/500
0s - loss: 3.2120 - acc: 0.0435
Epoch 9/500
0s - loss: 3.1994 - acc: 0.0435
Epoch 10/500
0s - loss: 3.1884 - acc: 0.0435
Epoch 11/500
0s - loss: 3.1756 - acc: 0.0435
Epoch 12/500
0s - loss: 3.1622 - acc: 0.0435
Epoch 13/500
0s - loss: 3.1445 - acc: 0.0435
Epoch 14/500
0s - loss: 3.1309 - acc: 0.0435
Epoch 15/500
0s - loss: 3.1143 - acc: 0.0435
Epoch 16/500
0s - loss: 3.0971 - acc: 0.0435
Epoch 17/500
0s - loss: 3.0796 - acc: 0.0000e+00
Epoch 18/500
0s - loss: 3.0630 - acc: 0.0435
Epoch 19/500
0s - loss: 3.0465 - acc: 0.0435
Epoch 20/500
0s - loss: 3.0294 - acc: 0.0435
Epoch 21/500
0s - loss: 3.0111 - acc: 0.0435
Epoch 22/500
0s - loss: 2.9954 - acc: 0.0000e+00
Epoch 23/50

0s - loss: 1.0286 - acc: 0.8696
Epoch 182/500
0s - loss: 1.0180 - acc: 0.8696
Epoch 183/500
0s - loss: 1.0292 - acc: 0.8696
Epoch 184/500
0s - loss: 1.0176 - acc: 0.8696
Epoch 185/500
0s - loss: 1.0081 - acc: 0.8696
Epoch 186/500
0s - loss: 1.0055 - acc: 0.8696
Epoch 187/500
0s - loss: 0.9970 - acc: 0.8696
Epoch 188/500
0s - loss: 0.9958 - acc: 0.9130
Epoch 189/500
0s - loss: 0.9902 - acc: 0.8261
Epoch 190/500
0s - loss: 0.9864 - acc: 0.9130
Epoch 191/500
0s - loss: 0.9813 - acc: 0.8696
Epoch 192/500
0s - loss: 0.9717 - acc: 0.8696
Epoch 193/500
0s - loss: 0.9683 - acc: 0.9565
Epoch 194/500
0s - loss: 0.9607 - acc: 0.9130
Epoch 195/500
0s - loss: 0.9535 - acc: 0.9130
Epoch 196/500
0s - loss: 0.9502 - acc: 0.8696
Epoch 197/500
0s - loss: 0.9481 - acc: 0.9565
Epoch 198/500
0s - loss: 0.9510 - acc: 0.9130
Epoch 199/500
0s - loss: 0.9391 - acc: 0.9130
Epoch 200/500
0s - loss: 0.9326 - acc: 0.9130
Epoch 201/500
0s - loss: 0.9323 - acc: 0.8696
Epoch 202/500
0s - loss: 0.9268 - acc: 0.9130
Ep

0s - loss: 0.4060 - acc: 1.0000
Epoch 361/500
0s - loss: 0.3926 - acc: 0.9565
Epoch 362/500
0s - loss: 0.3903 - acc: 0.9565
Epoch 363/500
0s - loss: 0.3858 - acc: 0.9565
Epoch 364/500
0s - loss: 0.3912 - acc: 0.9565
Epoch 365/500
0s - loss: 0.3892 - acc: 0.9565
Epoch 366/500
0s - loss: 0.3819 - acc: 0.9565
Epoch 367/500
0s - loss: 0.3835 - acc: 1.0000
Epoch 368/500
0s - loss: 0.3785 - acc: 1.0000
Epoch 369/500
0s - loss: 0.3794 - acc: 0.9565
Epoch 370/500
0s - loss: 0.3737 - acc: 0.9565
Epoch 371/500
0s - loss: 0.3703 - acc: 1.0000
Epoch 372/500
0s - loss: 0.3719 - acc: 0.9565
Epoch 373/500
0s - loss: 0.3646 - acc: 1.0000
Epoch 374/500
0s - loss: 0.3693 - acc: 1.0000
Epoch 375/500
0s - loss: 0.3657 - acc: 1.0000
Epoch 376/500
0s - loss: 0.3656 - acc: 0.9565
Epoch 377/500
0s - loss: 0.3617 - acc: 1.0000
Epoch 378/500
0s - loss: 0.3617 - acc: 1.0000
Epoch 379/500
0s - loss: 0.3587 - acc: 0.9565
Epoch 380/500
0s - loss: 0.3580 - acc: 0.9565
Epoch 381/500
0s - loss: 0.3532 - acc: 0.9565
Ep