In [1]:
from sklearn.svm import SVC
from sklearn.neural_network import MLPClassifier
from sklearn.naive_bayes import BernoulliNB
from sklearn.model_selection import cross_val_score
from sklearn.preprocessing import normalize
from py.utils import load_data
import pickle

directory = '../data/'
heads = ['l30_r15', 'l10_r10', 'l5_r5']
n_cv = 5

In [2]:
import pickle
performances = {}

for head in heads:
    print('\n\nhead = %s' % head)
    x, y, x_words, vocabs = load_data(head, directory)
    x = normalize(x)
    
    classifier = BernoulliNB()
    scores = cross_val_score(classifier, x, y, cv=n_cv)
    print('\nBernoulli Naive Bayes: ', end='')
    print(' > %s' % ['%.5f' % s for s in scores])
    performances[('BernoulliNB norm', head)] = scores
    with open('performance_other_classifier norm.pkl', 'wb') as f:
        pickle.dump(performances, f)
    classifier.fit(x, y)
    model_name = 'BernoulliNB norm ' + head
    with open('../models/%s.pkl' % model_name, 'wb') as f:
        pickle.dump(classifier, f)   
        
    
    classifier = MLPClassifier(hidden_layer_sizes=(5,))
    scores = cross_val_score(classifier, x, y, cv=n_cv)
    print('Multilayer Perceptron Classifier (h=[5]): ', end='')
    print(' > %s' % ['%.5f' % s for s in scores])
    performances[('MLPClassifier (5,) norm', head)] = scores
    with open('performance_other_classifier norm.pkl', 'wb') as f:
        pickle.dump(performances, f)
    classifier.fit(x, y)
    model_name = 'MLPClassifier (5,) norm' + head
    with open('../models/%s.pkl' % model_name, 'wb') as f:
        pickle.dump(classifier, f)   

    
    classifier = MLPClassifier(hidden_layer_sizes=(20,))
    scores = cross_val_score(classifier, x, y, cv=n_cv)
    print('Multilayer Perceptron Classifier (h=[20])', end='')
    print(' > %s' % ['%.5f' % s for s in scores])
    performances[('MLPClassifier (20,) norm', head)] = scores
    with open('performance_other_classifier norm.pkl', 'wb') as f:
        pickle.dump(performances, f)
    classifier.fit(x, y)
    model_name = 'MLPClassifier (20,) norm' + head
    with open('../models/%s.pkl' % model_name, 'wb') as f:
        pickle.dump(classifier, f)   

    
    classifier = MLPClassifier(hidden_layer_sizes=(50,10))
    scores = cross_val_score(classifier, x, y, cv=n_cv)
    print('Multilayer Perceptron Classifier (h=[50, 10]): ', end='')
    print(' > %s' % ['%.5f' % s for s in scores])
    performances[('MLPClassifier (50,10) norm', head)] = scores
    with open('performance_other_classifier norm.pkl', 'wb') as f:
        pickle.dump(performances, f)
    classifier.fit(x, y)
    model_name = 'MLPClassifier (50,10) norm' + head
    with open('../models/%s.pkl' % model_name, 'wb') as f:
        pickle.dump(classifier, f)   

    
    classifier = SVC(C=10.0, kernel='rbf',shrinking=True)
    scores = cross_val_score(classifier, x, y, cv=n_cv)
    print('Support Vector Machine (rbf, C=10.0): ', end='')
    print(' > %s' % ['%.5f' % s for s in scores])
    performances[('SVC (C=10) norm', head)] = scores
    with open('performance_other_classifier norm.pkl', 'wb') as f:
        pickle.dump(performances, f)
    classifier.fit(x, y)
    model_name = 'Support Vector Machine (rbf, C=10.0) norm' + head
    with open('../models/%s.pkl' % model_name, 'wb') as f:
        pickle.dump(classifier, f)   

    
    classifier = SVC(C=1.0, kernel='rbf',shrinking=True)
    scores = cross_val_score(classifier, x, y, cv=n_cv)
    print('Support Vector Machine (rbf, C=1.0): ', end='')
    print(' > %s' % ['%.5f' % s for s in scores])
    performances[('SVC (C=1.0) norm', head)] = scores
    with open('performance_other_classifier norm.pkl', 'wb') as f:
        pickle.dump(performances, f)
    classifier.fit(x, y)
    model_name = 'SVC (C=1.0) norm' + head
    with open('../models/%s.pkl' % model_name, 'wb') as f:
        pickle.dump(classifier, f)   

    
    classifier = SVC(C=0.1, kernel='rbf',shrinking=True)
    scores = cross_val_score(classifier, x, y, cv=n_cv)
    print('Support Vector Machine (rbf, C=0.1): ', end='')
    print(' > %s' % ['%.5f' % s for s in scores])
    performances[('SVC (C=0.1) norm', head)] = scores
    with open('performance_other_classifier norm.pkl', 'wb') as f:
        pickle.dump(performances, f)
    classifier.fit(x, y)
    model_name = 'SVC (C=0.1) norm' + head
    with open('../models/%s.pkl' % model_name, 'wb') as f:
        pickle.dump(classifier, f)   

    
    print('-' * 80)



head = l30_r15
x shape = (15105, 2770)
y shape = (15105,)
# features = 2770
# L words = 15105

Bernoulli Naive Bayes:  > ['0.99371', '0.99470', '0.99503', '0.99305', '0.99338']




Multilayer Perceptron Classifier (h=[5]):  > ['0.99636', '0.99735', '0.99768', '0.99669', '0.99570']
Multilayer Perceptron Classifier (h=[20]) > ['0.99636', '0.99735', '0.99801', '0.99603', '0.99536']
Multilayer Perceptron Classifier (h=[50, 10]):  > ['0.99636', '0.99669', '0.99735', '0.99702', '0.99636']
Support Vector Machine (rbf, C=10.0):  > ['0.99140', '0.98908', '0.98643', '0.98643', '0.98477']
Support Vector Machine (rbf, C=1.0):  > ['0.83388', '0.83416', '0.83416', '0.83416', '0.83411']
Support Vector Machine (rbf, C=0.1):  > ['0.83388', '0.83416', '0.83416', '0.83416', '0.83411']
--------------------------------------------------------------------------------


head = l10_r10
x shape = (31544, 3515)
y shape = (31544,)
# features = 3515
# L words = 31544

Bernoulli Naive Bayes:  > ['0.99097', '0.98954', '0.99065', '0.99049', '0.98763']




Multilayer Perceptron Classifier (h=[5]):  > ['0.99746', '0.99604', '0.99461', '0.99604', '0.99588']
Multilayer Perceptron Classifier (h=[20]) > ['0.99746', '0.99604', '0.99477', '0.99588', '0.99620']
Multilayer Perceptron Classifier (h=[50, 10]):  > ['0.99731', '0.99572', '0.99445', '0.99572', '0.99620']
Support Vector Machine (rbf, C=10.0):  > ['0.98780', '0.98605', '0.98637', '0.98447', '0.98716']
Support Vector Machine (rbf, C=1.0):  > ['0.86908', '0.86908', '0.86908', '0.86908', '0.86906']
Support Vector Machine (rbf, C=0.1):  > ['0.86908', '0.86908', '0.86908', '0.86908', '0.86906']
--------------------------------------------------------------------------------


head = l5_r5
x shape = (50229, 5361)
y shape = (50229,)
# features = 5361
# L words = 50229

Bernoulli Naive Bayes:  > ['0.98507', '0.98507', '0.98557', '0.98756', '0.98606']




Multilayer Perceptron Classifier (h=[5]):  > ['0.99602', '0.99462', '0.99572', '0.99602', '0.99522']
Multilayer Perceptron Classifier (h=[20]) > ['0.99572', '0.99423', '0.99562', '0.99582', '0.99482']
Multilayer Perceptron Classifier (h=[50, 10]):  > ['0.99612', '0.99462', '0.99502', '0.99582', '0.99482']
Support Vector Machine (rbf, C=10.0):  > ['0.97472', '0.97193', '0.97521', '0.97651', '0.97362']
Support Vector Machine (rbf, C=1.0):  > ['0.88981', '0.88981', '0.88981', '0.88981', '0.88990']
Support Vector Machine (rbf, C=0.1):  > ['0.88981', '0.88981', '0.88981', '0.88981', '0.88990']
--------------------------------------------------------------------------------
