## import data from dataset using pandas
* data is tab seperated
* split to label and text

In [5]:
import pandas as pd

df = pd.read_table('smsspamcollection/SMSSpamCollection',
                  sep='\t',
                  header=None,
                  names=['label','sms_message'])
df.head()

Unnamed: 0,label,sms_message
0,ham,"Go until jurong point, crazy.. Available only ..."
1,ham,Ok lar... Joking wif u oni...
2,spam,Free entry in 2 a wkly comp to win FA Cup fina...
3,ham,U dun say so early hor... U c already then say...
4,ham,"Nah I don't think he goes to usf, he lives aro..."


## Data processing
mapped all texts from Ham to 0 and spam as 1  
printed out number of rows and columns

In [6]:
df['label'] = df.label.map({'ham':0,'spam':1})
print(df.shape)
df.head()

(5572, 2)


Unnamed: 0,label,sms_message
0,0,"Go until jurong point, crazy.. Available only ..."
1,0,Ok lar... Joking wif u oni...
2,1,Free entry in 2 a wkly comp to win FA Cup fina...
3,0,U dun say so early hor... U c already then say...
4,0,"Nah I don't think he goes to usf, he lives aro..."


## Splitting data into testing and training sets 
X_train and X_test are sms_messages are the data  
Y_train and Y_test are labels  

In [7]:
from sklearn.cross_validation import train_test_split

X_train, X_test, y_train, y_test = train_test_split(df['sms_message'],
                                                   df['label'],
                                                   random_state=1)

print('Number of rows = {}'.format(df.shape[0]))
print('Number of rows in training set = {}'.format(X_train.shape[0]))
print('Number of rows in testign set = {}'.format(X_test.shape[0]))

Number of rows = 5572
Number of rows in training set = 4179
Number of rows in testign set = 1393


## Applying bags of words to dataset

In [9]:
from sklearn.feature_extraction.text import CountVectorizer
count_vector = CountVectorizer()

training_data = count_vector.fit_transform(X_train)
testing_data = count_vector.transform(X_test)

# doc_array = count_vector.transform(X_train).toarray()
# print(doc_array)

# frequency_matrix = pd.DataFrame(doc_array, 
#                                 columns = count_vector.get_feature_names())
# print(frequency_matrix)

## Applying Naive Bayes using Scikit-Learn 

In [10]:
from sklearn.naive_bayes import MultinomialNB
naive_bayes = MultinomialNB()
naive_bayes.fit(training_data,y_train)

predictions = naive_bayes.predict(testing_data)

## Evaluating Model
* Accuracy  : Correct/total
* Precision : TP/(TP+FP)
* Recall    : TP/(TP+FP)
* F1 score

In [11]:
from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score
print('Accuracy  = {}'.format(accuracy_score(y_test,predictions)))
print('Precision = {}'.format(precision_score(y_test,predictions)))
print('Recall    = {}'.format(recall_score(y_test,predictions)))
print('F1 Score  = {}'.format(f1_score(y_test,predictions)))

Accuracy  = 0.988513998564
Precision = 0.972067039106
Recall    = 0.940540540541
F1 Score  = 0.956043956044


### END 
___