Importing Libraries

In [1]:
import numpy as np
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import accuracy_score
from sklearn.preprocessing import LabelEncoder

Data pre processing

In [2]:
raw_mail_data = pd.read_csv("mail_data.csv")

In [3]:
mail_data = raw_mail_data.where((pd.notnull(raw_mail_data)), '')
mail_data.shape

(5572, 2)

In [4]:
X = mail_data['Message']
y = mail_data['Category']
print(X.shape, y.shape)
print(y.head())

(5572,) (5572,)
0     ham
1     ham
2    spam
3     ham
4     ham
Name: Category, dtype: object


Label Encoding

In [5]:
labels = LabelEncoder()

encodedtarget = labels.fit_transform(y)
encodedtarget

array([0, 0, 1, ..., 0, 0, 0])

Split

In [6]:
X_train, X_test, Y_train, Y_test = train_test_split(X, encodedtarget, test_size=0.2, random_state=3)

In [7]:
feature_extraction = TfidfVectorizer(min_df = 1, stop_words='english', lowercase=True)
X_train_features = feature_extraction.fit_transform(X_train)
X_test_features = feature_extraction.transform(X_test)

In [8]:
lr = LogisticRegression()
lr.fit(X_train_features, Y_train)

Accuracy

In [9]:
prediction_on_training_data = lr.predict(X_train_features)
accuracy_on_training_data = accuracy_score(Y_train, prediction_on_training_data)
print('Accuracy on training data : ', accuracy_on_training_data)

Accuracy on training data :  0.9670181736594121


In [10]:
prediction_on_test_data = lr.predict(X_test_features)
accuracy_on_test_data = accuracy_score(Y_test, prediction_on_test_data)
print('Accuracy on test data : ', accuracy_on_test_data)

Accuracy on test data :  0.9659192825112107


Prediction System

In [None]:
input_mail = ["I've been searching for the right words to thank you for this breather. I promise i wont take your help for granted and will fulfil my promise. You have been wonderful and a blessing at all times"]

input_data_features = feature_extraction.transform(input_mail)

prediction = lr.predict(input_data_features)
print("Prediction : ", prediction)
print("Prediction : ", labels.inverse_transform(prediction))


if (prediction[0] == 0):
  print('Ham mail')
else:
  print('Spam mail')

Prediction :  [0]
Prediction :  ['ham']
Ham mail


Pickle files for GUI

In [12]:
import pickle
pickle.dump(feature_extraction, open('vector.pkl', 'wb'))
pickle.dump(lr, open('model.pkl', 'wb'))