# Load Raw Data and Clean

In [64]:
import string
import re
from pickle import load
from pickle import dump
from unicodedata import normalize
from numpy import array
from numpy.random import rand
from numpy.random import shuffle

In [65]:
def load_doc(filename):
    # open the file as read only
    file = open(filename, mode='rt', encoding='utf-8')
    # read all text
    text = file.read()
    # close the file
    file.close()
    return text

In [66]:
# split a loaded document into sentences
def to_pairs(doc):
    lines = doc.strip().split('\n')
    pairs = [line.split('\t') for line in  lines]
    return pairs

In [67]:
# clean a list of lines
def clean_pairs(lines):
    cleaned = list()
    # prepare regex for char filtering
    re_print = re.compile('[^%s]' % re.escape(string.printable))
    # prepare translation table for removing punctuation
    table = str.maketrans('', '', string.punctuation)
    for pair in lines:
        clean_pair = list()
        for line in pair:
            # normalize unicode characters
            line = normalize('NFD', line).encode('ascii', 'ignore')
            line = line.decode('UTF-8')
            # tokenize on white space
            line = line.split()
            # convert to lowercase
            line = [word.lower() for word in line]
            # remove punctuation from each token
            line = [word.translate(table) for word in line]
            # remove non-printable chars form each token
            line = [re_print.sub('', w) for w in line]
            # remove tokens with numbers in them
            line = [word for word in line if word.isalpha()]
            # store as string
            clean_pair.append(' '.join(line))
        cleaned.append(clean_pair)
    return array(cleaned)

In [68]:
# save a list of clean sentences to file
def save_clean_data(sentences, filename):
    dump(sentences, open(filename, 'wb'))
    print('Saved: %s' % filename)

In [69]:
# load dataset
filename = 'deu.txt'
doc = load_doc(filename)

In [70]:
# split into english-german pairs
pairs = to_pairs(doc)

In [71]:
# clean sentences
clean_pairs = clean_pairs(pairs)

In [72]:
# save clean pairs to file
save_clean_data(clean_pairs, 'english-german.pkl')

Saved: english-german.pkl


In [73]:
for i in range(5):
    print('[%s] => [%s]' % (clean_pairs[i,0], clean_pairs[i,1]))

[go] => [geh]
[hi] => [hallo]
[hi] => [gru gott]
[run] => [lauf]
[run] => [lauf]


In [18]:
# load a clean dataset
def load_clean_sentences(filename):
    return load(open(filename, 'rb'))

In [19]:
raw_dataset = load_clean_sentences('english-german.pkl')

In [21]:
print(len(raw_dataset))

221533


In [20]:
n_sentences = 10000
dataset = raw_dataset[:n_sentences, :]

In [22]:
shuffle(dataset)

In [23]:
train, test = dataset[:9000], dataset[9000:]

In [24]:
save_clean_data(dataset, 'english-german-both.pkl')
save_clean_data(train, 'english-german-train.pkl')
save_clean_data(test, 'english-german-test.pkl')

Saved: english-german-both.pkl
Saved: english-german-train.pkl
Saved: english-german-test.pkl


# Build the NMT Model

In [25]:
from pickle import load
from numpy import array
from keras.preprocessing.text import Tokenizer
from keras.preprocessing.sequence import pad_sequences
from keras.utils import to_categorical
from keras.utils.vis_utils import plot_model
from keras.models import Sequential
from keras.layers import LSTM
from keras.layers import Dense
from keras.layers import Embedding
from keras.layers import RepeatVector
from keras.layers import TimeDistributed
from keras.callbacks import ModelCheckpoint

In [26]:
def create_tokenizer(lines):
    tokenizer = Tokenizer()
    tokenizer.fit_on_texts(lines)
    return tokenizer

In [27]:
# max sentence length
def max_length(lines):
    return max(len(line.split()) for line in lines)

In [28]:
# encode and pad sequences
def encode_sequences(tokenizer, length, lines):
    # integer encode sequences
    X = tokenizer.texts_to_sequences(lines)
    # pad sequences with 0 values
    X = pad_sequences(X, maxlen=length, padding='post')
    return X

In [29]:
# one hot encode target sequence
def encode_output(sequences, vocab_size):
    ylist = list()
    for sequence in sequences:
        encoded = to_categorical(sequence, num_classes=vocab_size)
        ylist.append(encoded)
    y = array(ylist)
    y = y.reshape(sequences.shape[0], sequences.shape[1], vocab_size)
    return y

In [30]:
# define NMT model
def define_model(src_vocab, tar_vocab, src_timesteps, tar_timesteps, n_units):
    model = Sequential()
    model.add(Embedding(src_vocab, n_units, input_length=src_timesteps, mask_zero=True))
    model.add(LSTM(n_units))
    model.add(RepeatVector(tar_timesteps))
    model.add(LSTM(n_units, return_sequences=True))
    model.add(TimeDistributed(Dense(tar_vocab, activation='softmax')))
    return model

In [31]:
dataset = load_clean_sentences('english-german-both.pkl')
train = load_clean_sentences('english-german-train.pkl')
test = load_clean_sentences('english-german-test.pkl')

In [32]:
# prepare english tokenizer
eng_tokenizer = create_tokenizer(dataset[:, 0])
eng_vocab_size = len(eng_tokenizer.word_index) + 1
eng_length = max_length(dataset[:, 0])
print('English Vocabulary Size: %d' % eng_vocab_size)
print('English Max Length: %d' % (eng_length))

English Vocabulary Size: 2241
English Max Length: 5


In [33]:
# prepare german tokenizer
ger_tokenizer = create_tokenizer(dataset[:, 1])
ger_vocab_size = len(ger_tokenizer.word_index) + 1
ger_length = max_length(dataset[:, 1])
print('German Vocabulary Size: %d' % ger_vocab_size)
print('German Max Length: %d' % (ger_length))

German Vocabulary Size: 3572
German Max Length: 9


In [34]:
# prepare training data
trainX = encode_sequences(ger_tokenizer, ger_length, train[:, 1])
trainY = encode_sequences(eng_tokenizer, eng_length, train[:, 0])
trainY = encode_output(trainY, eng_vocab_size)

In [35]:
# prepare validation data
testX = encode_sequences(ger_tokenizer, ger_length, test[:, 1])
testY = encode_sequences(eng_tokenizer, eng_length, test[:, 0])
testY = encode_output(testY, eng_vocab_size)

In [36]:
# define model
model = define_model(ger_vocab_size, eng_vocab_size, ger_length, eng_length, 256)
model.compile(optimizer='adam', loss='categorical_crossentropy')

In [38]:
# summarize defined model
print(model.summary())
plot_model(model, to_file='model.png', show_shapes=True)

Model: "sequential"
_________________________________________________________________
Layer (type)                 Output Shape              Param #   
embedding (Embedding)        (None, 9, 256)            914432    
_________________________________________________________________
lstm (LSTM)                  (None, 256)               525312    
_________________________________________________________________
repeat_vector (RepeatVector) (None, 5, 256)            0         
_________________________________________________________________
lstm_1 (LSTM)                (None, 5, 256)            525312    
_________________________________________________________________
time_distributed (TimeDistri (None, 5, 2241)           575937    
Total params: 2,540,993
Trainable params: 2,540,993
Non-trainable params: 0
_________________________________________________________________
None
('Failed to import pydot. You must `pip install pydot` and install graphviz (https://graphviz.gitlab.io/dow

In [39]:
# fit model
filename = 'model.h5'
checkpoint = ModelCheckpoint(filename, monitor='val_loss', verbose=1, save_best_only=True, mode='min')
model.fit(trainX, trainY, epochs=30, batch_size=64, validation_data=(testX, testY), callbacks=[checkpoint], verbose=2)

Epoch 1/30

Epoch 00001: val_loss improved from inf to 3.36899, saving model to model.h5
141/141 - 9s - loss: 4.1321 - val_loss: 3.3690
Epoch 2/30

Epoch 00002: val_loss improved from 3.36899 to 3.22993, saving model to model.h5
141/141 - 7s - loss: 3.2161 - val_loss: 3.2299
Epoch 3/30

Epoch 00003: val_loss improved from 3.22993 to 3.15028, saving model to model.h5
141/141 - 8s - loss: 3.0777 - val_loss: 3.1503
Epoch 4/30

Epoch 00004: val_loss improved from 3.15028 to 3.00750, saving model to model.h5
141/141 - 8s - loss: 2.9329 - val_loss: 3.0075
Epoch 5/30

Epoch 00005: val_loss improved from 3.00750 to 2.90200, saving model to model.h5
141/141 - 8s - loss: 2.7752 - val_loss: 2.9020
Epoch 6/30

Epoch 00006: val_loss improved from 2.90200 to 2.81506, saving model to model.h5
141/141 - 8s - loss: 2.6389 - val_loss: 2.8151
Epoch 7/30

Epoch 00007: val_loss improved from 2.81506 to 2.74327, saving model to model.h5
141/141 - 8s - loss: 2.5057 - val_loss: 2.7433
Epoch 8/30

Epoch 00008:

<tensorflow.python.keras.callbacks.History at 0x1ebe2f92880>

# Evaluate the Model

In [49]:
from pickle import load
from numpy import array
from numpy import argmax
from keras.preprocessing.text import Tokenizer
from keras.preprocessing.sequence import pad_sequences
from keras.models import load_model
from nltk.translate.bleu_score import corpus_bleu

In [54]:
# map an integer to a word
def word_for_id(integer, tokenizer):
    for word, index in tokenizer.word_index.items():
        if index == integer:
            return word
    return None

In [55]:
# generate target given source sequence
def predict_sequence(model, tokenizer, source):
    prediction = model.predict(source, verbose=0)[0]
    integers = [argmax(vector) for vector in prediction]
    target = list()
    for i in integers:
        word = word_for_id(i, tokenizer)
        if word is None:
            break
        target.append(word)
    return ' '.join(target)

In [59]:
def evaluate_model(model, tokenizer, sources, raw_dataset):
    actual, predicted = list(), list()
    for i, source in enumerate(sources):
        # translate encoded source text
        source = source.reshape((1, source.shape[0]))
        translation = predict_sequence(model, eng_tokenizer, source)
        raw_target, raw_src, _ = raw_dataset[i]
        if i < 10:
            print('src=[%s], target=[%s], predicted=[%s]' % (raw_src, raw_target, translation))
        actual.append([raw_target.split()])
        predicted.append(translation.split())
    # calculate BLEU score
    print('BLEU-1: %f' % corpus_bleu(actual, predicted, weights=(1.0, 0, 0, 0)))
    print('BLEU-2: %f' % corpus_bleu(actual, predicted, weights=(0.5, 0.5, 0, 0)))
    print('BLEU-3: %f' % corpus_bleu(actual, predicted, weights=(0.3, 0.3, 0.3, 0)))
    print('BLEU-4: %f' % corpus_bleu(actual, predicted, weights=(0.25, 0.25, 0.25, 0.25)))


In [50]:
# load datasets
dataset = load_clean_sentences('english-german-both.pkl')
train = load_clean_sentences('english-german-train.pkl')
test = load_clean_sentences('english-german-test.pkl')

In [51]:
# prepare english tokenizer
eng_tokenizer = create_tokenizer(dataset[:, 0])
eng_vocab_size = len(eng_tokenizer.word_index) + 1
eng_length = max_length(dataset[:, 0])

In [52]:
# prepare german tokenizer
ger_tokenizer = create_tokenizer(dataset[:, 1])
ger_vocab_size = len(ger_tokenizer.word_index) + 1
ger_length = max_length(dataset[:, 1])

In [53]:
# prepare data
trainX = encode_sequences(ger_tokenizer, ger_length, train[:, 1])
testX = encode_sequences(ger_tokenizer, ger_length, test[:, 1])

In [57]:
# load model
model = load_model('model.h5')

In [62]:
# test on some training sequences
print('train')
evaluate_model(model, eng_tokenizer, trainX, train)

train
src=[tom fluchte], target=[tom swore], predicted=[tom swore]
src=[er ist abgeschlossen], target=[its locked], predicted=[its locked]
src=[wie kommt das], target=[why is that], predicted=[how is that]
src=[ich muss mich weigern], target=[i must refuse], predicted=[i must refuse]
src=[hast du angerufen], target=[you called], predicted=[you called]
src=[das hat mumm erfordert], target=[that took guts], predicted=[that took guts]
src=[habe ich recht], target=[am i right], predicted=[am i correct]
src=[ich bin jetzt ein mann], target=[im a man now], predicted=[im a man]
src=[das ist wichtig], target=[its important], predicted=[its important]
src=[das passiert], target=[it happens], predicted=[it happens]
BLEU-1: 0.873667
BLEU-2: 0.821140
BLEU-3: 0.715909
BLEU-4: 0.393324


In [63]:
# test on some test sequences
print('test')
evaluate_model(model, eng_tokenizer, testX, test)

test
src=[die rechnung bitte], target=[check please], predicted=[stop please]
src=[ich druckte sie], target=[i hugged her], predicted=[i get you]
src=[das ist in ordnung], target=[this is ok], predicted=[thats ok]
src=[weinen sie nicht], target=[dont cry], predicted=[dont cry]
src=[ich will einen beweis], target=[i want proof], predicted=[i want a job]
src=[ich werde ihn umbringen], target=[ill kill him], predicted=[ill get him]
src=[wir mussen gehorchen], target=[we must obey], predicted=[we can be]
src=[ich habe einen kuli], target=[i have a pen], predicted=[i have a dream]
src=[er lugt], target=[hes lying], predicted=[he is lying]
src=[iss dein essen], target=[eat your food], predicted=[eat your food]
BLEU-1: 0.542946
BLEU-2: 0.410680
BLEU-3: 0.319601
BLEU-4: 0.132113
