# Importing the Dataset
- Download Dataset [Click here](https://archive.ics.uci.edu/ml/datasets/sms+spam+collection "SMS Spam Collection Data Set")

In [1]:
import pandas as pd

In [2]:
messages = pd.read_csv('smsspamcollection/SMSSpamCollection', sep='\t',names=["label", "message"])

In [3]:
messages.head()

Unnamed: 0,label,message
0,ham,"Go until jurong point, crazy.. Available only ..."
1,ham,Ok lar... Joking wif u oni...
2,spam,Free entry in 2 a wkly comp to win FA Cup fina...
3,ham,U dun say so early hor... U c already then say...
4,ham,"Nah I don't think he goes to usf, he lives aro..."


## Data cleaning and preprocessing

In [4]:
import re
import nltk
from nltk.corpus import stopwords
from nltk.stem.porter import PorterStemmer

In [5]:
ps = PorterStemmer()

In [6]:
corpus = []
for i in range(0, len(messages)):
    review = re.sub('[^a-zA-Z]', ' ', messages['message'][i])
    review = review.lower()
    review = review.split()
    review = [ps.stem(word) for word in review if not word in stopwords.words('english')]
    review = ' '.join(review)
    corpus.append(review)

## Creating the Bag of Words model

In [7]:
from sklearn.feature_extraction.text import CountVectorizer
cv = CountVectorizer(max_features=2500)
X = cv.fit_transform(corpus).toarray()
y=pd.get_dummies(messages['label'])
y=y.iloc[:,1].values

In [8]:
print(X.shape,y.shape)

(5572, 2500) (5572,)


## Train Test Split

In [9]:
from sklearn.model_selection import train_test_split
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.20, random_state = 0)

## Training model using Naive bayes classifier

In [10]:
from sklearn.naive_bayes import MultinomialNB
spam_detect_model = MultinomialNB().fit(X_train, y_train)

In [11]:
y_pred=spam_detect_model.predict(X_test)

## Checking Accuracy

In [12]:
from sklearn.metrics import accuracy_score

In [13]:
print(accuracy_score(y_test, y_pred))

0.9856502242152466


## Confusion Matrix

In [14]:
from sklearn.metrics import confusion_matrix

In [15]:
print(confusion_matrix(y_test, y_pred))

[[946   9]
 [  7 153]]


## Creating the TF-IDF model

In [16]:
from sklearn.feature_extraction.text import TfidfVectorizer
tfidf = TfidfVectorizer()
X_tfidf = tfidf.fit_transform(corpus).toarray()

## Train Test Split for TF-IDF Model

In [17]:
X_train_tfidf, X_test_tfidf, y_train_tfidf, y_test_tfidf = train_test_split(X_tfidf, y, test_size = 0.20, random_state = 0)

In [18]:
print(X_train_tfidf.shape,X_test_tfidf.shape,y_train_tfidf.shape,y_test_tfidf.shape)

(4457, 6296) (1115, 6296) (4457,) (1115,)


## Training TF-IDF model using Naive bayes classifier

In [19]:
spam_detect_model_tfidf = MultinomialNB().fit(X_train_tfidf, y_train_tfidf)

In [20]:
y_pred_tfidf=spam_detect_model_tfidf.predict(X_test_tfidf)

## TF-IDF Accuracy 

In [21]:
print(accuracy_score(y_test_tfidf, y_pred_tfidf))

0.9695067264573991


## TF-IDF Confusion Matrix

In [22]:
print(confusion_matrix(y_test_tfidf, y_pred_tfidf))

[[955   0]
 [ 34 126]]
