##### A stemmed CountVectorizer

In [3]:
import copy
from sklearn.feature_extraction.text import CountVectorizer

# Overide CountVectorizer to integrate stemming
class StemmedCountVectorizer(CountVectorizer):
    def __init__(self, stemmer, **kwargs):
        super(StemmedCountVectorizer, self).__init__(**kwargs)
        self.stemmer = stemmer

    def build_analyzer(self):
        analyzer = super(StemmedCountVectorizer, self).build_analyzer()
        return lambda doc: (self.stemmer.stem(w) for w in analyzer(doc))

    def get_params(self, deep=True):
        params = super().get_params(deep=deep)
        cp = copy.copy(self)
        cp.__class__ = CountVectorizer
        params.update(CountVectorizer.get_params(cp, deep))
        return params

In [4]:
# -*- coding: utf-8 -*-

"""FakeNewsClassifier Trainer

This script conduct the following steps (overview):

1. Load the processes dataset from {PROJECT_DIR}/data/processed as dataframe
2. Concatenate title and text in one column
3. Split the dataset into train and test

4. Set up a sklearn pipeline with preprocessor and classifier for the combined column (title+text)
   and perform hyperparam search 
5. Extract best score and best_params for and log it
6. Predict on testset and log the score

7. Pickle best pipe to disk (final model)

"""
# %%
import logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger()

import os
import pandas as pd
from sklearn.pipeline import Pipeline
from sklearn.compose import ColumnTransformer
from sklearn.ensemble import VotingClassifier
from catboost import CatBoostClassifier
from sklearn.model_selection import train_test_split, GridSearchCV
from sklearn.metrics import classification_report, confusion_matrix, f1_score, recall_score, precision_score
from dotenv import find_dotenv, load_dotenv
import nltk
from nltk.stem.snowball import GermanStemmer
from sklearn.feature_extraction.text import CountVectorizer
import dagshub
from joblib import dump


load_dotenv(find_dotenv())
#%% Load and split dataset
# INPUTFILE = os.path.join(os.getenv('PROJECT_DIR'), 'data', 'processed', 'fake_news_processed.csv')
INPUTFILE = os.path.join('data', 'datasets_merged.csv')
df = pd.read_csv(INPUTFILE, sep=';')
logger.info('Distribution of fake news in entire dataset: \n%s' % df.fake.value_counts(normalize=True))

X = df['title']+' '+df['text']
y = df['fake']

X_train, X_test, y_train, y_test = train_test_split(X, y,
                    test_size=0.2,
                    stratify=y,
                    random_state=42)

# Download and init stopwordlist
nltk.download('stopwords')
STOPWORD_LIST = nltk.corpus.stopwords.words('german')

#print(X_train[100])

# %% Test with CountVectorizer
"""
count_vec = StemmedCountVectorizer(token_pattern=r'\b[a-zA-Z]{2,}\b',
                                max_features=100,
                                stop_words=STOPWORD_LIST,
                                lowercase=True,
                                stemmer=GermanStemmer(ignore_stopwords=True),
                                ngram_range=(2,2)
                        )

word_counts = count_vec.fit_transform(X_train).todense()

df_body = pd.DataFrame(word_counts, columns=count_vec.get_feature_names())
df_body_trans = pd.DataFrame(df_body.T.sum(axis=1), columns=['count'])
df_body_trans.sort_values(by='count', ascending=False).head(20)

"""

#%% Define param grid for tuning
param_grid = {'vectorizer__max_features':[500],
        'clf__n_estimators': [400],
        'clf__learning_rate': [0.9],
        'clf__max_depth': [5]}

# Constants for training
# MODEL_PATH = os.getenv('MODEL_PATH')
MODEL_PATH = 'model.pkl'
CV_SCORING = 'f1'
CV_FOLDS = 3

# %%
with dagshub.dagshub_logger() as dagslog:
        # Pipeline
        pipe = Pipeline(steps=[
        ('vectorizer', StemmedCountVectorizer(stop_words=STOPWORD_LIST,
                                   token_pattern=r'\b[a-zA-Z]{2,}\b',
                                   stemmer=GermanStemmer(ignore_stopwords=True),
                                   lowercase=True)),
        ('clf', CatBoostClassifier(allow_writing_files=False))
        ])
        
        # pipe_title_tune.get_params().keys()
        logger.info('Start training estimator pipeline')
        cv_grid = GridSearchCV(pipe, param_grid, scoring=CV_SCORING, cv=CV_FOLDS, n_jobs=2)
        cv_grid.fit(X_train, y_train)
        #cv_grid.best_estimator_.named_steps['clf'].get_all_params()

        # Log and assign best score and params to var
        logger.info('Best params from GridSearchCV: %s' % str(cv_grid.best_params_))
        dagslog.log_hyperparams({'model_path': MODEL_PATH})
        dagslog.log_hyperparams({'cv_score': CV_SCORING})
        dagslog.log_hyperparams({'best_cv_params': cv_grid.best_params_})

        logger.info('Best %s from GridSearchCV: %s' % (CV_SCORING, str(cv_grid.best_score_)))
        dagslog.log_metrics({'best_cv_score': cv_grid.best_score_})
        
        # Predict on testdata
        y_pred = cv_grid.best_estimator_.predict(X_test)
        logger.info('f1_score for testdata: %s' % str(f1_score(y_test, y_pred)))
        dagslog.log_metrics({'f1_score_on_testdata': f1_score(y_test, y_pred)})
        logger.info('precision_score for testdata: %s' % str(precision_score(y_test, y_pred)))
        dagslog.log_metrics({'precision_score_on_testdata': precision_score(y_test, y_pred)})
        logger.info('recall_score for testdata: %s' % str(recall_score(y_test, y_pred)))
        dagslog.log_metrics({'recall_score_on_testdata': recall_score(y_test, y_pred)})
        logger.info('Classification report for testdata: \n%s' % classification_report(y_test, y_pred))
        logger.info('Confusion matrix for testdata: \n%s' % confusion_matrix(y_test, y_pred))

        # Dump model to disk
        dump(cv_grid.best_estimator_, MODEL_PATH)

INFO:root:Distribution of fake news in entire dataset: 
0    0.922928
1    0.077072
Name: fake, dtype: float64
[nltk_data] Downloading package stopwords to
[nltk_data]     C:\Users\fjun\AppData\Roaming\nltk_data...
[nltk_data]   Package stopwords is already up-to-date!
INFO:root:Start training estimator pipeline


0:	learn: 0.1790348	total: 185ms	remaining: 1m 13s
1:	learn: 0.1292450	total: 250ms	remaining: 49.7s
2:	learn: 0.1143598	total: 308ms	remaining: 40.8s
3:	learn: 0.1028553	total: 359ms	remaining: 35.5s
4:	learn: 0.0949345	total: 412ms	remaining: 32.5s
5:	learn: 0.0891566	total: 481ms	remaining: 31.6s
6:	learn: 0.0860460	total: 547ms	remaining: 30.7s
7:	learn: 0.0828053	total: 599ms	remaining: 29.3s
8:	learn: 0.0804890	total: 669ms	remaining: 29.1s
9:	learn: 0.0802442	total: 719ms	remaining: 28s
10:	learn: 0.0773813	total: 776ms	remaining: 27.4s
11:	learn: 0.0743588	total: 847ms	remaining: 27.4s
12:	learn: 0.0742056	total: 885ms	remaining: 26.3s
13:	learn: 0.0712867	total: 952ms	remaining: 26.2s
14:	learn: 0.0710548	total: 988ms	remaining: 25.4s
15:	learn: 0.0693764	total: 1.04s	remaining: 24.9s
16:	learn: 0.0692207	total: 1.08s	remaining: 24.3s
17:	learn: 0.0672085	total: 1.13s	remaining: 24.1s
18:	learn: 0.0662156	total: 1.18s	remaining: 23.7s
19:	learn: 0.0660709	total: 1.22s	remainin

161:	learn: 0.0282877	total: 14.3s	remaining: 21.1s
162:	learn: 0.0282484	total: 14.4s	remaining: 21s
163:	learn: 0.0281843	total: 14.5s	remaining: 20.9s
164:	learn: 0.0280915	total: 14.8s	remaining: 21.1s
165:	learn: 0.0280339	total: 15s	remaining: 21.1s
166:	learn: 0.0279555	total: 15.1s	remaining: 21.1s
167:	learn: 0.0278377	total: 15.2s	remaining: 21s
168:	learn: 0.0278304	total: 15.4s	remaining: 21.1s
169:	learn: 0.0278144	total: 15.5s	remaining: 21s
170:	learn: 0.0277465	total: 15.6s	remaining: 21s
171:	learn: 0.0277411	total: 15.7s	remaining: 20.9s
172:	learn: 0.0276537	total: 15.9s	remaining: 20.9s
173:	learn: 0.0276439	total: 16s	remaining: 20.8s
174:	learn: 0.0272863	total: 16.2s	remaining: 20.9s
175:	learn: 0.0272144	total: 16.5s	remaining: 20.9s
176:	learn: 0.0272099	total: 16.8s	remaining: 21.2s
177:	learn: 0.0270794	total: 17.3s	remaining: 21.5s
178:	learn: 0.0270130	total: 17.7s	remaining: 21.8s
179:	learn: 0.0270002	total: 18s	remaining: 21.9s
180:	learn: 0.0269826	tota

320:	learn: 0.0178290	total: 28.7s	remaining: 7.07s
321:	learn: 0.0178107	total: 28.8s	remaining: 6.97s
322:	learn: 0.0177303	total: 28.8s	remaining: 6.87s
323:	learn: 0.0176133	total: 28.9s	remaining: 6.77s
324:	learn: 0.0175565	total: 28.9s	remaining: 6.67s
325:	learn: 0.0175428	total: 28.9s	remaining: 6.57s
326:	learn: 0.0174896	total: 29s	remaining: 6.47s
327:	learn: 0.0174893	total: 29s	remaining: 6.37s
328:	learn: 0.0174874	total: 29s	remaining: 6.27s
329:	learn: 0.0174652	total: 29.1s	remaining: 6.17s
330:	learn: 0.0174231	total: 29.1s	remaining: 6.07s
331:	learn: 0.0173994	total: 29.2s	remaining: 5.97s
332:	learn: 0.0173747	total: 29.2s	remaining: 5.88s
333:	learn: 0.0173731	total: 29.2s	remaining: 5.78s
334:	learn: 0.0172833	total: 29.3s	remaining: 5.69s
335:	learn: 0.0172530	total: 29.5s	remaining: 5.61s
336:	learn: 0.0170780	total: 29.6s	remaining: 5.53s
337:	learn: 0.0170560	total: 29.7s	remaining: 5.45s
338:	learn: 0.0170001	total: 29.8s	remaining: 5.37s
339:	learn: 0.0169

INFO:root:Best params from GridSearchCV: {'clf__learning_rate': 0.9, 'clf__max_depth': 5, 'clf__n_estimators': 400, 'vectorizer__max_features': 500}
INFO:root:Best f1 from GridSearchCV: 0.8555504998353983


398:	learn: 0.0147036	total: 34.5s	remaining: 86.5ms
399:	learn: 0.0146894	total: 34.6s	remaining: 0us


INFO:root:f1_score for testdata: 0.8681434599156118
INFO:root:precision_score for testdata: 0.8878101402373247
INFO:root:recall_score for testdata: 0.849329205366357
INFO:root:Classification report for testdata: 
              precision    recall  f1-score   support

           0       0.99      0.99      0.99     11599
           1       0.89      0.85      0.87       969

    accuracy                           0.98     12568
   macro avg       0.94      0.92      0.93     12568
weighted avg       0.98      0.98      0.98     12568

INFO:root:Confusion matrix for testdata: 
[[11495   104]
 [  146   823]]


In [9]:
key = 'HOME'
value = os.getenv(key)
  
# Print the value of 'HOME'
# environment variable
print("Value of 'HOME' environment variable :", value) 
  
# Get the value of 'JAVA_HOME'
# environment variable
key = 'MODEL_PATH'
value = os.getenv(key)
  
# Print the value of 'JAVA_HOME'
# environment variable
print("Value of 'JAVA_HOME' environment variable :", value) 

Value of 'HOME' environment variable : None
Value of 'JAVA_HOME' environment variable : None
