In [1]:
# things we need for NLP
import nltk
from nltk.stem.lancaster import LancasterStemmer
stemmer = LancasterStemmer()

# things we need for Tensorflow
import numpy as np
import tflearn
import tensorflow as tf
import random
import matplotlib.pyplot as plt

curses is not supported on this machine (please install/reinstall curses for an optimal experience)


In [2]:
# open intents file
import json
with open('intents.json') as json_data:
    intents = json.load(json_data)

In [3]:
words = []
classes = []
documents = []
ignore_words = ['?','!',]

# loop through each sentence in our intents patterns
for intent in intents['Intents']:
    for question in intent['Question']:
        # tokenize each word in the sentence
        w = nltk.word_tokenize(question)
        print(w)
        # add to our words list
        words.extend(w)
        # add to documents in our corpus
        documents.append((w, intent['Question']))
        # add to our classes list
        if intent['Question'] not in classes:
            classes.append(intent['Question'])
            


                        
# stem and lower each word and remove duplicates
words = [stemmer.stem(w.lower()) for w in words if w not in ignore_words]
words = sorted(list(set(words)))

# remove duplicates
#classes = sorted(list(set(classes)))
print (len(documents), "documents")
print (len(classes), "classes", classes)
print (len(words), "unique stemmed words", words)

['How', 'do', 'I', 'buy', 'online', '?']
['How', 'to', 'shop', 'online']
['Why', 'are', 'there', 'products', 'on', 'the', 'website', 'that', 'I', 'can', 'not', 'buy', 'online', '?']
['Are', 'your', 'prices', 'online', 'the', 'same', 'as', 'in', 'store', '?']
['What', 'if', 'I', 'change', 'my', 'mind', 'after', 'ordering', 'online', '?']
['Are', 'my', 'personal', 'details', 'given', 'to', 'third', 'parties', '?']
['Can', 'I', 'cancel', 'an', 'order', 'I', 'placed', 'online', '?']
['Can', 'I', 'add', 'to', 'or', 'change', 'an', 'existing', 'order', '?']
['Can', 'I', 'order', 'online', 'and', 'pick', 'up', 'from', 'a', 'store', '?']
9 documents
8 classes [['How do I buy online?', 'How to shop online'], ['Why are there products on the website that I cannot buy online?'], ['Are your prices online the same as in store?'], ['What if I change my mind after ordering online?'], ['Are my personal details given to third parties?'], ['Can I cancel an order I placed online?'], ['Can I add to or chan

In [4]:
# create our training data
training = []
output = []
# create an empty array for our output
output_empty = [0] * len(classes)

# training set, bag of words for each sentence
for doc in documents:
    # initialize our bag of words
    bag = []
    # list of tokenized words for the pattern
    pattern_words = doc[0]
    # stem each word
    pattern_words = [stemmer.stem(word.lower()) for word in pattern_words]
    # create our bag of words array
    for w in words:
        bag.append(1) if w in pattern_words else bag.append(0)

    # output is a '0' for each tag and '1' for current tag
    output_row = list(output_empty)
    output_row[classes.index(doc[1])] = 1

    training.append([bag, output_row])

# shuffle our features and turn into np.array
random.shuffle(training)
training = np.array(training)

# create train and test lists
train_x = list(training[:,0])
train_y = list(training[:,1])

x = np.random.randint(0, len(train_x))

print(train_x[x])
print(train_y[x])


[0, 1, 0, 1, 0, 0, 0, 0, 1, 0, 1, 0, 0, 1, 0, 0, 0, 1, 0, 0, 0, 0, 0, 0, 0, 1, 1, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 0, 1, 0, 0, 0, 0, 0]
[0, 0, 0, 0, 0, 0, 1, 0]


In [5]:
print(len(train_x))

9


In [6]:
# reset underlying graph data
tf.reset_default_graph()
# Build neural network
net = tflearn.input_data(shape=[None, len(train_x[0])])
net = tflearn.fully_connected(net, 32, activation="relu")
net = tflearn.fully_connected(net, 16, activation="relu")
net = tflearn.fully_connected(net, len(train_y[0]), activation='softmax')
net = tflearn.regression(net)

# Define model and setup tensorboard
model = tflearn.DNN(net, tensorboard_dir='tflearn_logs')
# Start training (apply gradient descent algorithm)
model.fit(train_x, train_y, n_epoch=1000, batch_size=4, show_metric=True)
    
model.save('model.tflearn')

Training Step: 2999  | total loss: [1m[32m0.00127[0m[0m | time: 0.008s
| Adam | epoch: 1000 | loss: 0.00127 - acc: 1.0000 -- iter: 8/9
Training Step: 3000  | total loss: [1m[32m0.00123[0m[0m | time: 0.019s
| Adam | epoch: 1000 | loss: 0.00123 - acc: 1.0000 -- iter: 9/9
--
INFO:tensorflow:C:\Users\XXX\Desktop\AI_model_0.1\model.tflearn is not in all_model_checkpoint_paths. Manually adding it.


In [7]:
def clean_up_sentence(sentence):
    # tokenize the pattern
    sentence_words = nltk.word_tokenize(sentence)
    # stem each word
    sentence_words = [stemmer.stem(word.lower()) for word in sentence_words]
    return sentence_words

# return bag of words array: 0 or 1 for each word in the bag that exists in the sentence
def bow(sentence, words, show_details=False):
    # tokenize the pattern
    sentence_words = clean_up_sentence(sentence)
    # bag of words
    bag = [0]*len(words)  
    for s in sentence_words:
        for i,w in enumerate(words):
            if w == s: 
                bag[i] = 1
                if show_details:
                    print ("found in bag: %s" % w)

    return(np.array(bag))

In [8]:
p = bow("I want to cancel my oder", words)
print (p)

[0 0 0 0 0 0 0 0 0 1 0 0 0 0 0 0 0 1 0 0 0 1 0 0 0 0 0 0 0 0 0 0 0 0 0 0 0
 0 0 0 1 0 0 0 0 0]


In [9]:
print(model.predict([p]))

[[ 0.0627869   0.01904533  0.0113242   0.67192173  0.04756602  0.16304596
   0.01551864  0.00879122]]


In [10]:
import pickle
pickle.dump( {'words':words, 'classes':classes, 'train_x':train_x, 'train_y':train_y}, open( "training_data", "wb" ) )