# Spam Detection

In [3]:
# import libraries
import string

## classical libraries
import pandas as pd
import numpy as np
import matplotlib.pyplot

## machine learning
import sklearn
from sklearn.feature_extraction.text import CountVectorizer
from sklearn.model_selection import train_test_split
from sklearn.ensemble import RandomForestClassifier

## NLP
import nltk
from nltk.corpus import stopwords
from nltk.stem.porter import PorterStemmer


In [4]:
## download important packages 
nltk.download('stopwords')

[nltk_data] Downloading package stopwords to /Users/ranu/nltk_data...
[nltk_data]   Package stopwords is already up-to-date!


True

In [5]:
df = pd.read_csv('spam_ham_dataset.csv')

In [6]:
df.head(5)

Unnamed: 0.1,Unnamed: 0,label,text,label_num
0,605,ham,Subject: enron methanol ; meter # : 988291\r\n...,0
1,2349,ham,"Subject: hpl nom for january 9 , 2001\r\n( see...",0
2,3624,ham,"Subject: neon retreat\r\nho ho ho , we ' re ar...",0
3,4685,spam,"Subject: photoshop , windows , office . cheap ...",1
4,2030,ham,Subject: re : indian springs\r\nthis deal is t...,0


In [7]:
## suppression des signes de ponctuation
df['text'] = df['text'].apply(lambda x: x.replace('\r\n', ' '))

In [8]:
df.info()

<class 'pandas.core.frame.DataFrame'>
RangeIndex: 5171 entries, 0 to 5170
Data columns (total 4 columns):
 #   Column      Non-Null Count  Dtype 
---  ------      --------------  ----- 
 0   Unnamed: 0  5171 non-null   int64 
 1   label       5171 non-null   object
 2   text        5171 non-null   object
 3   label_num   5171 non-null   int64 
dtypes: int64(2), object(2)
memory usage: 161.7+ KB


### Tokenization & Stemming

In [9]:
stemmer = PorterStemmer() # to reduce word
corpus = [] # create a corpus

stopwords_set = set(stopwords.words('english'))

for i in range(len(df)):
    ## conversion en minuscules
    text = df['text'].iloc[i].lower()
    ## tokenization
    text = text.translate(str.maketrans('', '', string.punctuation)).split()
    ## stemming
    text = [stemmer.stem(word) for word in text if word not in stopwords_set]
    text = ' '.join(text)
    corpus.append(text)

In [10]:
## difference between two text
df['text'].iloc[0]
corpus[0]

'subject enron methanol meter 988291 follow note gave monday 4 3 00 preliminari flow data provid daren pleas overrid pop daili volum present zero reflect daili activ obtain ga control chang need asap econom purpos'

### Vectorization

In [11]:
## vectorize
vectorizer = CountVectorizer()

## split data
X = vectorizer.fit_transform(corpus).toarray()
y = df['label_num']

X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)

In [12]:
## Classifier 
clf = RandomForestClassifier(n_jobs=-1) # number of jobs is negative here because we work on unstructured data

clf.fit(X_train, y_train)

clf.score(X_test, y_test)

0.9690821256038648

In [16]:
email_to_classify = df.text.values[10]
email_to_classify

"Subject: vocable % rnd - word asceticism vcsc - brand new stock for your attention vocalscape inc - the stock symbol is : vcsc vcsc will be our top stock pick for the month of april - stock expected to bounce to 12 cents level the stock hit its all time low and will bounce back stock is going to explode in next 5 days - watch it soar watch the stock go crazy this and next week . breaking news - vocalscape inc . announces agreement to resell mix network services current price : $ 0 . 025 we expect projected speculative price in next 5 days : $ 0 . 12 we expect projected speculative price in next 15 days : $ 0 . 15 vocalscape networks inc . is building a company that ' s revolutionizing the telecommunications industry with the most affordable phone systems , hardware , online software , and rates in canada and the us . vocalscape , a company with global reach , is receiving international attention for the development of voice over ip ( voip ) application solutions , including the award 

In [17]:
email_text = email_to_classify.lower().translate(str.maketrans('', '', string.punctuation)).split()
email_text = [stemmer.stem(word) for word in text if word not in stopwords_set]
email_text = ' '.join(email_text)

email_corpus = [email_text]

X_email = vectorizer.transform(email_corpus)

In [19]:
## Test

print(clf.predict(X_email), '\n')
print(df['label_num'].iloc[10])

[1] 

1
