In [31]:
# things we need for NLP
import nltk
from nltk.stem.lancaster import LancasterStemmer
stemmer = LancasterStemmer()

# things we need for Tensorflow
import numpy as np
import tflearn
import tensorflow as tf
import random

In [32]:
# import our chat-bot intents file
import json
with open('intents.json') as json_data:
    intents = json.load(json_data)
    
print(intents)

{u'intents': [{u'patterns': [u'Hi', u'How are you', u'Is anyone there?', u'Hello', u'Good day'], u'tag': u'greeting', u'responses': [u'Hello, thanks for visiting', u'Good to see you again', u'Hi there, how can I help?'], u'context_set': u''}, {u'patterns': [u'Bye', u'See you later', u'Goodbye'], u'tag': u'goodbye', u'responses': [u'See you later, thanks for visiting', u'Have a nice day', u'Bye! Come back again soon.']}, {u'patterns': [u'Tervist', u'Tere', u'Hei', u'Tsau'], u'tag': u'tere', u'responses': [u'Tere']}, {u'patterns': [u'What hours are you open?', u'What are your hours?', u'When are you open?'], u'tag': u'hours', u'responses': [u"We're open every day 9am-9pm", u'Our hours are 9am-9pm every day']}, {u'patterns': [u'Which mopeds do you have?', u'What kinds of mopeds are there?', u'What do you rent?'], u'tag': u'mopeds', u'responses': [u'We rent Yamaha, Piaggio and Vespa mopeds', u'We have Piaggio, Vespa and Yamaha mopeds']}, {u'patterns': [u'Do you take credit cards?', u'Do yo

In [33]:
words = []
classes = []
documents = []
ignore_words = ['?']
# loop through each sentence in our intents patterns
for intent in intents['intents']:
    for pattern in intent['patterns']:
        # tokenize each word in the sentence
        w = nltk.word_tokenize(pattern)
        # add to our words list
        words.extend(w)
        # add to documents in our corpus
        documents.append((w, intent['tag']))
        # add to our classes list
        if intent['tag'] not in classes:
            classes.append(intent['tag'])

# stem and lower each word and remove duplicates
words = [stemmer.stem(w.lower()) for w in words if w not in ignore_words]
words = sorted(list(set(words)))

# remove duplicates
classes = sorted(list(set(classes)))

print (len(documents), "documents")
print (len(classes), "classes", classes)
print (len(words), "unique stemmed words", words)

(28, 'documents')
(9, 'classes', [u'goodbye', u'greeting', u'hours', u'mopeds', u'opentoday', u'payments', u'rental', u'tere', u'today'])
(48, 'unique stemmed words', [u"'d", u'a', u'acceiv', u'anyon', u'ar', u'bye', u'can', u'card', u'cash', u'credit', u'day', u'do', u'doe', u'good', u'goodby', u'hav', u'hei', u'hello', u'hi', u'hour', u'how', u'i', u'is', u'kind', u'lat', u'lik', u'mastercard', u'mop', u'of', u'on', u'op', u'rent', u'see', u'tak', u'ter', u'terv', u'ther', u'thi', u'to', u'today', u'tsau', u'we', u'what', u'when', u'which', u'work', u'yo', u'you'])


In [34]:
# create our training data
training = []
output = []
# create an empty array for our output
output_empty = [0] * len(classes)

# training set, bag of words for each sentence
for doc in documents:
    # initialize our bag of words
    bag = []
    # list of tokenized words for the pattern
    pattern_words = doc[0]
    # stem each word
    pattern_words = [stemmer.stem(word.lower()) for word in pattern_words]
    # create our bag of words array
    for w in words:
        bag.append(1) if w in pattern_words else bag.append(0)

    # output is a '0' for each tag and '1' for current tag
    output_row = list(output_empty)
    output_row[classes.index(doc[1])] = 1

    training.append([bag, output_row])

# shuffle our features and turn into np.array
random.shuffle(training)
training = np.array(training)

# create train and test lists
train_x = list(training[:,0])
train_y = list(training[:,1])

In [56]:
# reset underlying graph data
tf.reset_default_graph()
# Build neural network
net = tflearn.input_data(shape=[None, len(train_x[0])])
net = tflearn.fully_connected(net, 8)
net = tflearn.fully_connected(net, 8)
net = tflearn.fully_connected(net, len(train_y[0]), activation='softmax')
net = tflearn.regression(net)

# Define model and setup tensorboard
model = tflearn.DNN(net, tensorboard_dir='tflearn_logs')
# Start training (apply gradient descent algorithm)
model.fit(train_x, train_y, n_epoch=1000, batch_size=8, show_metric=True)
model.save('model.tflearn')

Training Step: 3999  | total loss: [1m[32m1.17558[0m[0m | time: 0.008s
| Adam | epoch: 1000 | loss: 1.17558 - acc: 0.8646 -- iter: 24/28
Training Step: 4000  | total loss: [1m[32m1.09914[0m[0m | time: 0.012s
| Adam | epoch: 1000 | loss: 1.09914 - acc: 0.8781 -- iter: 28/28
--
INFO:tensorflow:/home/salts/cybersecurity-exhibition/chatbot/model.tflearn is not in all_model_checkpoint_paths. Manually adding it.


In [57]:
def clean_up_sentence(sentence):
    # tokenize the pattern
    sentence_words = nltk.word_tokenize(sentence)
    # stem each word
    sentence_words = [stemmer.stem(word.lower()) for word in sentence_words]
    return sentence_words

# return bag of words array: 0 or 1 for each word in the bag that exists in the sentence
def bow(sentence, words, show_details=False):
    # tokenize the pattern
    sentence_words = clean_up_sentence(sentence)
    # bag of words
    bag = [0]*len(words)  
    for s in sentence_words:
        for i,w in enumerate(words):
            if w == s: 
                bag[i] = 1
                if show_details:
                    print ("found in bag: %s" % w)

    return(np.array(bag))

In [58]:
p = bow("is your shop open today?", words)
print (p)
print (classes)

[0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0 1 0 0 0 0 0 0 0 1 0 0 0 0 0 0
 0 0 1 0 0 0 0 0 0 1 0]
[u'goodbye', u'greeting', u'hours', u'mopeds', u'opentoday', u'payments', u'rental', u'tere', u'today']


In [59]:
print(model.predict([p]))

[[3.0680597e-03 3.1782012e-02 6.5399848e-02 2.6260319e-03 3.8724738e-01
  4.3037400e-01 6.4717780e-05 3.6211096e-04 7.9075895e-02]]


In [60]:
# save all of our data structures
import pickle
pickle.dump( {'words':words, 'classes':classes, 'train_x':train_x, 'train_y':train_y}, open( "training_data", "wb" ) )