In [1]:
from sklearn import svm
from sklearn.svm import SVC 
import numpy as np
from sklearn.model_selection import GridSearchCV 
from save_csv import results_to_csv
from evaluation import classification_accuracy

In [2]:
full_mnist_data = np.load("../data/mnist-data.npz")
mnist_X = full_mnist_data["training_data"]
mnist_X = mnist_X.reshape(mnist_X.shape[0], -1)
mnist_y = full_mnist_data["training_labels"]
mnist_X = mnist_X[:500, :]
mnist_y = mnist_y[:500]
mnist_X.shape, mnist_y.shape

((500, 784), (500,))

In [3]:
param_grid = {'C': [0.1 * (10 ** power) for power in range(5)],  
              'gamma': [1, 0.1, 0.01, 0.001], 
              'kernel': ['rbf']}
grid = GridSearchCV(SVC(), param_grid, verbose = 5) 

In [4]:
grid.fit(mnist_X, mnist_y)

Fitting 5 folds for each of 20 candidates, totalling 100 fits
[CV 1/5] END ........C=0.1, gamma=1, kernel=rbf;, score=0.120 total time=   0.1s
[CV 2/5] END ........C=0.1, gamma=1, kernel=rbf;, score=0.120 total time=   0.1s
[CV 3/5] END ........C=0.1, gamma=1, kernel=rbf;, score=0.120 total time=   0.1s
[CV 4/5] END ........C=0.1, gamma=1, kernel=rbf;, score=0.120 total time=   0.1s
[CV 5/5] END ........C=0.1, gamma=1, kernel=rbf;, score=0.110 total time=   0.1s
[CV 1/5] END ......C=0.1, gamma=0.1, kernel=rbf;, score=0.120 total time=   0.1s
[CV 2/5] END ......C=0.1, gamma=0.1, kernel=rbf;, score=0.120 total time=   0.1s
[CV 3/5] END ......C=0.1, gamma=0.1, kernel=rbf;, score=0.120 total time=   0.1s
[CV 4/5] END ......C=0.1, gamma=0.1, kernel=rbf;, score=0.120 total time=   0.1s
[CV 5/5] END ......C=0.1, gamma=0.1, kernel=rbf;, score=0.110 total time=   0.1s
[CV 1/5] END .....C=0.1, gamma=0.01, kernel=rbf;, score=0.120 total time=   0.1s
[CV 2/5] END .....C=0.1, gamma=0.01, kernel=rbf

In [17]:
# best hyperparameters from fast gridsearch gamma = 1, C = 0.1
mnist_X = np.load("../data/mnist_training_data.npy")
mnist_y = mnist_X[:, -1]
mnist_X = mnist_X[:, :-1]

mnist_test_X = full_mnist_data["test_data"]
mnist_test_X = mnist_test_X.reshape(mnist_test_X.shape[0], -1)

clf = svm.SVC(C=0.1)
clf.fit(mnist_X, mnist_y)
predictions = clf.predict(mnist_test_X)
results_to_csv(predictions, "mnist_predictions")

In [12]:
predictions = clf.predict(mnist_X)
acc = classification_accuracy(predictions=predictions, true_labels=mnist_y)


In [16]:
np.unique(mnist_X, return_counts=True)

(array([  0,   1,   2,   3,   4,   5,   6,   7,   8,   9,  10,  11,  12,
         13,  14,  15,  16,  17,  18,  19,  20,  21,  22,  23,  24,  25,
         26,  27,  28,  29,  30,  31,  32,  33,  34,  35,  36,  37,  38,
         39,  40,  41,  42,  43,  44,  45,  46,  47,  48,  49,  50,  51,
         52,  53,  54,  55,  56,  57,  58,  59,  60,  61,  62,  63,  64,
         65,  66,  67,  68,  69,  70,  71,  72,  73,  74,  75,  76,  77,
         78,  79,  80,  81,  82,  83,  84,  85,  86,  87,  88,  89,  90,
         91,  92,  93,  94,  95,  96,  97,  98,  99, 100, 101, 102, 103,
        104, 105, 106, 107, 108, 109, 110, 111, 112, 113, 114, 115, 116,
        117, 118, 119, 120, 121, 122, 123, 124, 125, 126, 127, 128, 129,
        130, 131, 132, 133, 134, 135, 136, 137, 138, 139, 140, 141, 142,
        143, 144, 145, 146, 147, 148, 149, 150, 151, 152, 153, 154, 155,
        156, 157, 158, 159, 160, 161, 162, 163, 164, 165, 166, 167, 168,
        169, 170, 171, 172, 173, 174, 175, 176, 177

In [3]:
spam_full_data = np.load("../data/spam-data.npz")
spam_X = spam_full_data["training_data"]
spam_y = spam_full_data["training_labels"]

In [4]:
spam_X.shape, spam_y.shape

((4171, 32), (4171,))

In [5]:
param_grid = {'C': [0.1 * (10 ** power) for power in range(4)], 'kernel': ["linear"]}
grid = GridSearchCV(SVC(), param_grid, verbose = 5) 
grid.fit(spam_X, spam_y)

Fitting 5 folds for each of 4 candidates, totalling 20 fits
[CV 1/5] END ..............C=0.1, kernel=linear;, score=0.794 total time=   0.1s
[CV 2/5] END ..............C=0.1, kernel=linear;, score=0.800 total time=   0.1s
[CV 3/5] END ..............C=0.1, kernel=linear;, score=0.797 total time=   0.1s
[CV 4/5] END ..............C=0.1, kernel=linear;, score=0.789 total time=   0.1s
[CV 5/5] END ..............C=0.1, kernel=linear;, score=0.800 total time=   0.1s
[CV 1/5] END ..............C=1.0, kernel=linear;, score=0.804 total time=   0.7s
[CV 2/5] END ..............C=1.0, kernel=linear;, score=0.803 total time=   0.6s
[CV 3/5] END ..............C=1.0, kernel=linear;, score=0.800 total time=   0.6s
[CV 4/5] END ..............C=1.0, kernel=linear;, score=0.788 total time=   0.4s
[CV 5/5] END ..............C=1.0, kernel=linear;, score=0.811 total time=   0.6s
[CV 1/5] END .............C=10.0, kernel=linear;, score=0.804 total time=   1.4s
[CV 2/5] END .............C=10.0, kernel=linear;,

In [10]:
train = np.load("../data/spam_training_data.npy")
train_X = train[:, :-1]
train_y = train[:, -1]
test = np.load("../data/spam_testing_data.npy")
test_X = train[:, :-1]
test_y = train[:, -1]

In [11]:
clf = svm.SVC(C=100)
clf.fit(train_X, train_y)
prediction = clf.predict(test_X)
accuracy = classification_accuracy(predictions=prediction, true_labels=test_y)
accuracy

0.8519628408750375

In [12]:
clf = svm.SVC(C=100)
clf.fit(spam_X, spam_y)
prediction = clf.predict(test_X)
accuracy = classification_accuracy(predictions=prediction, true_labels=test_y)

0.8555588852262511

In [16]:
kaggle_X = spam_full_data["test_data"]
prediction = clf.predict(kaggle_X)
results_to_csv(y_test=prediction, file_name="spam_predictions.csv")
