##### A stemmed CountVectorizer

In [11]:
import copy
from sklearn.feature_extraction.text import CountVectorizer

# Overide CountVectorizer to integrate stemming
class StemmedCountVectorizer(CountVectorizer):
    def __init__(self, stemmer, **kwargs):
        super(StemmedCountVectorizer, self).__init__(**kwargs)
        self.stemmer = stemmer

    def build_analyzer(self):
        analyzer = super(StemmedCountVectorizer, self).build_analyzer()
        return lambda doc: (self.stemmer.stem(w) for w in analyzer(doc))

    def get_params(self, deep=True):
        params = super().get_params(deep=deep)
        cp = copy.copy(self)
        cp.__class__ = CountVectorizer
        params.update(CountVectorizer.get_params(cp, deep))
        return params

In [16]:
# -*- coding: utf-8 -*-

"""FakeNewsClassifier Trainer

This script conduct the following steps (overview):

1. Load the processes dataset from {PROJECT_DIR}/data/processed as dataframe
2. Concatenate title and text in one column
3. Split the dataset into train and test

4. Set up a sklearn pipeline with preprocessor and classifier for the combined column (title+text)
   and perform hyperparam search 
5. Extract best score and best_params for and log it
6. Predict on testset and log the score

7. Pickle best pipe to disk (final model)

"""
# %%
import logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger()

import os
import pandas as pd
from sklearn.pipeline import Pipeline
from sklearn.compose import ColumnTransformer
from sklearn.ensemble import VotingClassifier
from catboost import CatBoostClassifier, Pool
from sklearn.model_selection import train_test_split, GridSearchCV
from sklearn.metrics import classification_report, confusion_matrix, f1_score, recall_score, precision_score
from dotenv import find_dotenv, load_dotenv
import nltk
from nltk.stem.snowball import GermanStemmer
from sklearn.feature_extraction.text import CountVectorizer
import dagshub
from joblib import dump


load_dotenv(find_dotenv())
#%% Load and split dataset
# INPUTFILE = os.path.join(os.getenv('PROJECT_DIR'), 'data', 'processed', 'fake_news_processed.csv')
INPUTFILE = os.path.join('data', 'BuzzFeed-Webis.csv')
df = pd.read_csv(INPUTFILE, sep=',', keep_default_na=False)
logger.info('Distribution of fake news in entire dataset: \n%s' % df.veracity.value_counts(normalize=True))


# # Convert NaN titles to an empty string for concatination of title and text
# df[['title']] = df[['title']].fillna('')

print(df['title'].isnull().values.any())
print(df['mainText'].isnull().values.any())
print(df['veracity'].isnull().values.any())

X = df['title']+' '+df['mainText']
y = df['veracity']



X_train, X_test, y_train, y_test = train_test_split(X, y,
                    test_size=0.2,
                    stratify=y,
                    random_state=42)

# Download and init stopwordlist
nltk.download('stopwords')
STOPWORD_LIST = nltk.corpus.stopwords.words('english')

#print(X_train[100])

# %% Test with CountVectorizer
"""
count_vec = StemmedCountVectorizer(token_pattern=r'\b[a-zA-Z]{2,}\b',
                                max_features=100,
                                stop_words=STOPWORD_LIST,
                                lowercase=True,
                                stemmer=GermanStemmer(ignore_stopwords=True),
                                ngram_range=(2,2)
                        )

word_counts = count_vec.fit_transform(X_train).todense()

df_body = pd.DataFrame(word_counts, columns=count_vec.get_feature_names())
df_body_trans = pd.DataFrame(df_body.T.sum(axis=1), columns=['count'])
df_body_trans.sort_values(by='count', ascending=False).head(20)

"""

#%% Define param grid for tuning
param_grid = {'vectorizer__max_features':[500],
        'clf__n_estimators': [400],
        'clf__learning_rate': [0.9],
        'clf__max_depth': [5]}

# Constants for training
# MODEL_PATH = os.getenv('MODEL_PATH')
MODEL_PATH = 'model.pkl'
CV_SCORING = 'f1'
CV_FOLDS = 3

# %%
with dagshub.dagshub_logger() as dagslog:
        # Pipeline
        pipe = Pipeline(steps=[
        ('vectorizer', StemmedCountVectorizer(stop_words=STOPWORD_LIST,
                                   token_pattern=r'\b[a-zA-Z]{2,}\b',
                                   stemmer=GermanStemmer(ignore_stopwords=True),
                                   lowercase=True)),
        ('clf', CatBoostClassifier(allow_writing_files=False))
        ])
        
        # pipe_title_tune.get_params().keys()
        logger.info('Start training estimator pipeline')
        cv_grid = GridSearchCV(pipe, param_grid, scoring=CV_SCORING, cv=CV_FOLDS, n_jobs=2)
        cv_grid.fit(X_train, y_train)
        #cv_grid.best_estimator_.named_steps['clf'].get_all_params()

        # Log and assign best score and params to var
        logger.info('Best params from GridSearchCV: %s' % str(cv_grid.best_params_))
        dagslog.log_hyperparams({'model_path': MODEL_PATH})
        dagslog.log_hyperparams({'cv_score': CV_SCORING})
        dagslog.log_hyperparams({'best_cv_params': cv_grid.best_params_})

        logger.info('Best %s from GridSearchCV: %s' % (CV_SCORING, str(cv_grid.best_score_)))
        dagslog.log_metrics({'best_cv_score': cv_grid.best_score_})
        
        # Predict on testdata
        y_pred = cv_grid.best_estimator_.predict(X_test)
        logger.info('f1_score for testdata: %s' % str(f1_score(y_test, y_pred)))
        dagslog.log_metrics({'f1_score_on_testdata': f1_score(y_test, y_pred)})
        logger.info('precision_score for testdata: %s' % str(precision_score(y_test, y_pred)))
        dagslog.log_metrics({'precision_score_on_testdata': precision_score(y_test, y_pred)})
        logger.info('recall_score for testdata: %s' % str(recall_score(y_test, y_pred)))
        dagslog.log_metrics({'recall_score_on_testdata': recall_score(y_test, y_pred)})
        logger.info('Classification report for testdata: \n%s' % classification_report(y_test, y_pred))
        logger.info('Confusion matrix for testdata: \n%s' % confusion_matrix(y_test, y_pred))

        # Dump model to disk
        dump(cv_grid.best_estimator_, MODEL_PATH)

INFO:root:Distribution of fake news in entire dataset: 
0    0.812296
1    0.187704
Name: veracity, dtype: float64
[nltk_data] Downloading package stopwords to
[nltk_data]     C:\Users\fjun\AppData\Roaming\nltk_data...
[nltk_data]   Package stopwords is already up-to-date!
INFO:root:Start training estimator pipeline


False
False
False
0:	learn: 0.4838447	total: 23.4ms	remaining: 9.34s
1:	learn: 0.3818877	total: 26.3ms	remaining: 5.24s
2:	learn: 0.3493742	total: 29.1ms	remaining: 3.84s
3:	learn: 0.3313220	total: 31.8ms	remaining: 3.15s
4:	learn: 0.3141586	total: 34.3ms	remaining: 2.71s
5:	learn: 0.3086892	total: 36.8ms	remaining: 2.42s
6:	learn: 0.3049385	total: 39.4ms	remaining: 2.21s
7:	learn: 0.3011920	total: 41.9ms	remaining: 2.05s
8:	learn: 0.2847506	total: 44.7ms	remaining: 1.94s
9:	learn: 0.2582713	total: 47.2ms	remaining: 1.84s
10:	learn: 0.2368846	total: 49.9ms	remaining: 1.76s
11:	learn: 0.2315861	total: 52.5ms	remaining: 1.7s
12:	learn: 0.2292199	total: 55.2ms	remaining: 1.64s
13:	learn: 0.2121081	total: 57.8ms	remaining: 1.59s
14:	learn: 0.1889032	total: 60.4ms	remaining: 1.55s
15:	learn: 0.1785348	total: 63ms	remaining: 1.51s
16:	learn: 0.1759440	total: 65.7ms	remaining: 1.48s
17:	learn: 0.1661175	total: 68.3ms	remaining: 1.45s
18:	learn: 0.1620314	total: 71ms	remaining: 1.42s
19:	learn

193:	learn: 0.0058813	total: 559ms	remaining: 594ms
194:	learn: 0.0058775	total: 562ms	remaining: 591ms
195:	learn: 0.0058555	total: 564ms	remaining: 587ms
196:	learn: 0.0058530	total: 567ms	remaining: 584ms
197:	learn: 0.0058141	total: 569ms	remaining: 581ms
198:	learn: 0.0057422	total: 572ms	remaining: 578ms
199:	learn: 0.0056579	total: 574ms	remaining: 574ms
200:	learn: 0.0056444	total: 577ms	remaining: 571ms
201:	learn: 0.0056370	total: 579ms	remaining: 568ms
202:	learn: 0.0056306	total: 582ms	remaining: 565ms
203:	learn: 0.0055331	total: 584ms	remaining: 562ms
204:	learn: 0.0055330	total: 587ms	remaining: 558ms
205:	learn: 0.0055329	total: 589ms	remaining: 555ms
206:	learn: 0.0055329	total: 592ms	remaining: 552ms
207:	learn: 0.0055325	total: 594ms	remaining: 548ms
208:	learn: 0.0054482	total: 597ms	remaining: 545ms
209:	learn: 0.0053653	total: 599ms	remaining: 542ms
210:	learn: 0.0052259	total: 602ms	remaining: 539ms
211:	learn: 0.0052194	total: 605ms	remaining: 536ms
212:	learn: 

INFO:root:Best params from GridSearchCV: {'clf__learning_rate': 0.9, 'clf__max_depth': 5, 'clf__n_estimators': 400, 'vectorizer__max_features': 500}
INFO:root:Best f1 from GridSearchCV: 0.4234074868221209


384:	learn: 0.0034275	total: 1.1s	remaining: 42.8ms
385:	learn: 0.0034275	total: 1.1s	remaining: 40ms
386:	learn: 0.0034274	total: 1.1s	remaining: 37.1ms
387:	learn: 0.0034274	total: 1.11s	remaining: 34.2ms
388:	learn: 0.0034270	total: 1.11s	remaining: 31.4ms
389:	learn: 0.0034260	total: 1.11s	remaining: 28.5ms
390:	learn: 0.0034260	total: 1.11s	remaining: 25.6ms
391:	learn: 0.0034248	total: 1.12s	remaining: 22.8ms
392:	learn: 0.0034247	total: 1.12s	remaining: 19.9ms
393:	learn: 0.0034246	total: 1.12s	remaining: 17.1ms
394:	learn: 0.0034246	total: 1.12s	remaining: 14.2ms
395:	learn: 0.0034245	total: 1.13s	remaining: 11.4ms
396:	learn: 0.0034244	total: 1.13s	remaining: 8.53ms
397:	learn: 0.0034245	total: 1.13s	remaining: 5.68ms
398:	learn: 0.0034244	total: 1.15s	remaining: 2.88ms
399:	learn: 0.0034244	total: 1.15s	remaining: 0us


INFO:root:f1_score for testdata: 0.3736263736263736
INFO:root:precision_score for testdata: 0.5
INFO:root:recall_score for testdata: 0.2982456140350877
INFO:root:Classification report for testdata: 
              precision    recall  f1-score   support

           0       0.85      0.93      0.89       249
           1       0.50      0.30      0.37        57

    accuracy                           0.81       306
   macro avg       0.68      0.61      0.63       306
weighted avg       0.79      0.81      0.79       306

INFO:root:Confusion matrix for testdata: 
[[232  17]
 [ 40  17]]


In [9]:
key = 'HOME'
value = os.getenv(key)
  
# Print the value of 'HOME'
# environment variable
print("Value of 'HOME' environment variable :", value) 
  
# Get the value of 'JAVA_HOME'
# environment variable
key = 'MODEL_PATH'
value = os.getenv(key)
  
# Print the value of 'JAVA_HOME'
# environment variable
print("Value of 'JAVA_HOME' environment variable :", value) 

Value of 'HOME' environment variable : None
Value of 'JAVA_HOME' environment variable : None
