In [1]:
# Naive LSTM to learn three-char window to one-char mapping
import numpy
from keras.models import Sequential
from keras.layers import Dense
from keras.layers import LSTM
from keras.utils import np_utils

Using TensorFlow backend.


In [2]:
# fix random seed for reproducibility
numpy.random.seed(7)

In [3]:
# define the raw dataset
alphabet = "ABCDEFGHIJKLMNOPQRSTUVWXYZ"
# create mapping of characters to integers (0-25) and the reverse
char_to_int = dict((c, i) for i, c in enumerate(alphabet))
int_to_char = dict((i, c) for i, c in enumerate(alphabet))
# prepare the dataset of input to output pairs encoded as integers
seq_length = 3
dataX = []
dataY = []
for i in range(0, len(alphabet) - seq_length, 1):
    seq_in = alphabet[i:i + seq_length]
    seq_out = alphabet[i + seq_length]
    dataX.append([char_to_int[char] for char in seq_in])
    dataY.append(char_to_int[seq_out])
    print(seq_in, '->', seq_out)

ABC -> D
BCD -> E
CDE -> F
DEF -> G
EFG -> H
FGH -> I
GHI -> J
HIJ -> K
IJK -> L
JKL -> M
KLM -> N
LMN -> O
MNO -> P
NOP -> Q
OPQ -> R
PQR -> S
QRS -> T
RST -> U
STU -> V
TUV -> W
UVW -> X
VWX -> Y
WXY -> Z


In [4]:
# reshape X to be [samples, time steps, features]
X = numpy.reshape(dataX, (len(dataX), 1, seq_length))
# normalize
X = X / float(len(alphabet))
# one hot encode the output variable
y = np_utils.to_categorical(dataY)

In [5]:
# create and fit the model
model = Sequential()
model.add(LSTM(32, input_shape=(X.shape[1], X.shape[2])))
model.add(Dense(y.shape[1], activation='softmax'))
model.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])
model.fit(X, y, epochs=500, batch_size=1, verbose=2)
# summarize performance of the model
scores = model.evaluate(X, y, verbose=0)
print("Model Accuracy: %.2f%%" % (scores[1]*100))
# demonstrate some model predictions
for pattern in dataX:
    x = numpy.reshape(pattern, (1, 1, len(pattern)))
    x = x / float(len(alphabet))
    prediction = model.predict(x, verbose=0)
    index = numpy.argmax(prediction)
    result = int_to_char[index]
    seq_in = [int_to_char[value] for value in pattern]
    print(seq_in, "->", result)

Epoch 1/500
0s - loss: 3.2651 - acc: 0.0000e+00
Epoch 2/500
0s - loss: 3.2527 - acc: 0.0435
Epoch 3/500
0s - loss: 3.2462 - acc: 0.0435
Epoch 4/500
0s - loss: 3.2402 - acc: 0.0000e+00
Epoch 5/500
0s - loss: 3.2339 - acc: 0.0435
Epoch 6/500
0s - loss: 3.2274 - acc: 0.0435
Epoch 7/500
0s - loss: 3.2209 - acc: 0.0435
Epoch 8/500
0s - loss: 3.2143 - acc: 0.0000e+00
Epoch 9/500
0s - loss: 3.2068 - acc: 0.0435
Epoch 10/500
0s - loss: 3.1993 - acc: 0.0435
Epoch 11/500
0s - loss: 3.1918 - acc: 0.0435
Epoch 12/500
0s - loss: 3.1839 - acc: 0.0000e+00
Epoch 13/500
0s - loss: 3.1757 - acc: 0.0435
Epoch 14/500
0s - loss: 3.1674 - acc: 0.0435
Epoch 15/500
0s - loss: 3.1586 - acc: 0.0000e+00
Epoch 16/500
0s - loss: 3.1499 - acc: 0.0435
Epoch 17/500
0s - loss: 3.1419 - acc: 0.0000e+00
Epoch 18/500
0s - loss: 3.1341 - acc: 0.0000e+00
Epoch 19/500
0s - loss: 3.1246 - acc: 0.0435
Epoch 20/500
0s - loss: 3.1168 - acc: 0.0435
Epoch 21/500
0s - loss: 3.1096 - acc: 0.0435
Epoch 22/500
0s - loss: 3.1018 - acc

0s - loss: 2.1416 - acc: 0.3913
Epoch 185/500
0s - loss: 2.1383 - acc: 0.3478
Epoch 186/500
0s - loss: 2.1366 - acc: 0.3478
Epoch 187/500
0s - loss: 2.1327 - acc: 0.3043
Epoch 188/500
0s - loss: 2.1316 - acc: 0.3043
Epoch 189/500
0s - loss: 2.1283 - acc: 0.3478
Epoch 190/500
0s - loss: 2.1242 - acc: 0.3478
Epoch 191/500
0s - loss: 2.1225 - acc: 0.3043
Epoch 192/500
0s - loss: 2.1178 - acc: 0.3043
Epoch 193/500
0s - loss: 2.1171 - acc: 0.2609
Epoch 194/500
0s - loss: 2.1140 - acc: 0.2174
Epoch 195/500
0s - loss: 2.1108 - acc: 0.3043
Epoch 196/500
0s - loss: 2.1099 - acc: 0.3478
Epoch 197/500
0s - loss: 2.1051 - acc: 0.3043
Epoch 198/500
0s - loss: 2.1025 - acc: 0.3478
Epoch 199/500
0s - loss: 2.1004 - acc: 0.3478
Epoch 200/500
0s - loss: 2.0981 - acc: 0.3478
Epoch 201/500
0s - loss: 2.0951 - acc: 0.3478
Epoch 202/500
0s - loss: 2.0926 - acc: 0.3043
Epoch 203/500
0s - loss: 2.0919 - acc: 0.3043
Epoch 204/500
0s - loss: 2.0875 - acc: 0.3478
Epoch 205/500
0s - loss: 2.0844 - acc: 0.3043
Ep

0s - loss: 1.8077 - acc: 0.6087
Epoch 365/500
0s - loss: 1.8069 - acc: 0.5652
Epoch 366/500
0s - loss: 1.8060 - acc: 0.6522
Epoch 367/500
0s - loss: 1.8042 - acc: 0.6087
Epoch 368/500
0s - loss: 1.8021 - acc: 0.6957
Epoch 369/500
0s - loss: 1.8003 - acc: 0.6957
Epoch 370/500
0s - loss: 1.8005 - acc: 0.6957
Epoch 371/500
0s - loss: 1.7980 - acc: 0.5652
Epoch 372/500
0s - loss: 1.7977 - acc: 0.6522
Epoch 373/500
0s - loss: 1.7946 - acc: 0.6957
Epoch 374/500
0s - loss: 1.7929 - acc: 0.6957
Epoch 375/500
0s - loss: 1.7939 - acc: 0.6957
Epoch 376/500
0s - loss: 1.7907 - acc: 0.6087
Epoch 377/500
0s - loss: 1.7892 - acc: 0.6522
Epoch 378/500
0s - loss: 1.7900 - acc: 0.6087
Epoch 379/500
0s - loss: 1.7862 - acc: 0.6522
Epoch 380/500
0s - loss: 1.7872 - acc: 0.6522
Epoch 381/500
0s - loss: 1.7871 - acc: 0.6087
Epoch 382/500
0s - loss: 1.7851 - acc: 0.7391
Epoch 383/500
0s - loss: 1.7812 - acc: 0.6957
Epoch 384/500
0s - loss: 1.7813 - acc: 0.6522
Epoch 385/500
0s - loss: 1.7825 - acc: 0.7391
Ep