In [1]:
def text2paragraphs(filename, min_size=1):
    """ A text contained in the file 'filename' will be read 
    and chopped into paragraphs.
    Paragraphs with a string length less than min_size will be ignored.
    A list of paragraph strings will be returned"""
    
    txt = open(filename).read()
    paragraphs = [para for para in txt.split("\n\n") if len(para) > min_size]
    return paragraphs


In [2]:
# the position of lables is very important
# it corresponds to a novel by that author within "files"
# the position of the author is also relevant, as it will correspond to metrics
# i.e. Samuel Butler's metrics are always returned in position 1
labels = ['Virginia Woolf', 'Samuel Butler', 'Herman Melville', 
          'David Herbert Lawrence', 'Daniel Defoe', 'James Joyce']


# names of books we have to train our machine model
files = ['night_and_day_virginia_woolf.txt', 'the_way_of_all_flash_butler.txt',
         'moby_dick_melville.txt', 'sons_and_lovers_lawrence.txt',
         'robinson_crusoe_defoe.txt', 'james_joyce_ulysses.txt']

# location of our books
path = "books/"


In [3]:
data = []
targets = []
counter = 0

# loop across all files we have downloaded
for fname in files:
    paras = text2paragraphs(path + fname, min_size=150) # return a book with paragraphs over 150 chars in a list
    data.extend(paras)
    targets += [counter] * len(paras)
    counter += 1


In [4]:
# cell is useless, because train_test_split will do the shuffling!

import random

data_targets = list(zip(data, targets))
# create random permutation on list:
data_targets = random.sample(data_targets, len(data_targets))

data, targets = list(zip(*data_targets))


In [5]:
from sklearn.model_selection import train_test_split

res = train_test_split(data, targets, 
                       train_size=0.8,
                       test_size=0.2,
                       random_state=42)
train_data, test_data, train_targets, test_targets = res 


In [6]:
from sklearn.feature_extraction.text import CountVectorizer, ENGLISH_STOP_WORDS

from sklearn.naive_bayes import MultinomialNB
from sklearn import metrics

vectorizer = CountVectorizer(stop_words=ENGLISH_STOP_WORDS)

vectors = vectorizer.fit_transform(train_data)

# creating a classifier
classifier = MultinomialNB(alpha=.01)
classifier.fit(vectors, train_targets)

vectors_test = vectorizer.transform(test_data)

predictions = classifier.predict(vectors_test)
accuracy_score = metrics.accuracy_score(test_targets, 
                                        predictions)
f1_score = metrics.f1_score(test_targets, 
                            predictions, 
                            average='macro')

print("accuracy score: ", accuracy_score)
print("F1-score: ", f1_score)


accuracy score:  0.9227000544365814
F1-score:  0.9178544256400646


In [7]:
# we want to use paragraphs from this 2nd Virginia Wolf 
paras = text2paragraphs(path + "the_voyage_out_virginia_woolf.txt", min_size=250)

# start on paragraph 100 and go to paragraph 500
first_para, last_para = 100, 500
vectors_test = vectorizer.transform(paras[first_para: last_para]) # pass a list of strings that will be used to make predictions against
#vectors_test = vectorizer.transform(["To be or not to be"])

predictions = classifier.predict(vectors_test) # make our predictions
print(predictions)
targets = [0] * (last_para - first_para)
accuracy_score = metrics.accuracy_score(targets, 
                                        predictions)
precision_score = metrics.precision_score(targets, 
                                          predictions, 
                                          average='macro')

f1_score = metrics.f1_score(targets, 
                            predictions, 
                            average='macro')

print("accuracy score: ", accuracy_score)
print("precision score: ", accuracy_score)
print("F1-score: ", f1_score)


[5 0 0 5 0 0 2 0 2 5 0 0 0 0 0 0 0 0 1 0 1 0 0 5 1 5 0 1 0 0 0 0 5 2 2 5 0
 2 2 5 0 0 0 0 0 3 0 0 0 0 2 0 2 5 2 0 0 0 0 1 0 0 5 3 3 2 0 0 0 0 5 2 5 0
 0 0 0 0 0 2 2 0 2 2 2 2 5 5 0 5 1 0 0 0 0 5 1 0 5 0 0 3 5 5 2 5 0 5 5 0 5
 0 0 0 3 0 0 3 2 2 0 0 5 0 1 5 2 2 5 1 0 5 0 0 0 0 0 5 1 5 0 0 0 0 5 2 0 0
 0 0 5 0 0 5 5 1 0 1 1 0 0 0 0 1 0 5 0 1 0 0 0 5 3 5 5 0 2 0 0 0 0 0 0 0 5
 5 0 0 0 0 0 0 0 0 0 0 0 1 0 0 0 0 1 5 5 0 0 0 5 5 5 2 0 5 0 5 3 0 0 0 5 0
 0 0 2 0 0 0 0 2 3 0 0 0 0 5 0 0 5 3 5 1 0 5 5 5 5 0 5 0 1 0 1 0 0 0 0 1 3
 1 1 0 5 5 5 5 2 5 0 0 0 5 0 2 2 0 1 0 0 0 0 0 0 0 0 4 0 2 0 0 1 0 0 0 0 1
 1 0 5 5 5 0 5 1 0 0 0 0 5 0 0 0 0 5 3 0 3 0 0 0 0 0 0 0 0 0 3 0 0 0 0 0 5
 3 3 5 0 3 0 0 0 1 0 1 0 0 3 0 2 0 3 0 0 1 1 0 0 0 0 0 3 0 0 0 2 2 2 3 0 0
 1 1 0 0 0 0 0 0 0 0 0 0 0 0 3 3 0 0 0 0 1 5 0 0 5 0 0 0 0 0]
accuracy score:  0.5875
precision score:  0.5875
F1-score:  0.12335958005249344


In [8]:
# perform a probability test
predictions = classifier.predict_proba(vectors_test)
print(predictions)


[[4.53996778e-005 3.30516175e-006 2.28281847e-007 2.56260328e-007
  2.63908128e-016 9.99950811e-001]
 [9.90942859e-001 1.87349064e-006 5.16175017e-003 2.25796438e-011
  1.00232845e-017 3.89351703e-003]
 [9.99870516e-001 2.52643506e-011 3.70172549e-014 5.04303810e-010
  3.21424032e-016 1.29483876e-004]
 ...
 [1.00000000e+000 3.61928906e-010 3.73007318e-032 5.73076833e-020
  3.61739448e-042 4.40613574e-032]
 [9.99998550e-001 1.45026975e-006 2.55592935e-020 2.24580195e-040
  9.81140334e-056 1.63018368e-014]
 [1.00000000e+000 1.70814440e-043 4.81585250e-065 4.47776733e-082
  2.05298749e-109 3.85623676e-059]]


In [9]:
for i in range(0, 10):
    print(predictions[i], paras[i+first_para])


[4.53996778e-05 3.30516175e-06 2.28281847e-07 2.56260328e-07
 2.63908128e-16 9.99950811e-01] "That's the painful thing about pets," said Mr. Dalloway; "they die. The
first sorrow I can remember was for the death of a dormouse. I regret to
say that I sat upon it. Still, that didn't make one any the less sorry.
Here lies the duck that Samuel Johnson sat on, eh? I was big for my
age."
[9.90942859e-01 1.87349064e-06 5.16175017e-03 2.25796438e-11
 1.00232845e-17 3.89351703e-03] "Please tell me--everything." That was what she wanted to say. He had
drawn apart one little chink and showed astonishing treasures. It seemed
to her incredible that a man like that should be willing to talk to her.
He had sisters and pets, and once lived in the country. She stirred her
tea round and round; the bubbles which swam and clustered in the cup
seemed to her like the union of their minds.
[9.99870516e-01 2.52643506e-11 3.70172549e-14 5.04303810e-10
 3.21424032e-16 1.29483876e-04] The talk meanwhile raced pa