##### A stemmed CountVectorizer

In [1]:
import copy
from sklearn.feature_extraction.text import CountVectorizer

# Overide CountVectorizer to integrate stemming
class StemmedCountVectorizer(CountVectorizer):
    def __init__(self, stemmer, **kwargs):
        super(StemmedCountVectorizer, self).__init__(**kwargs)
        self.stemmer = stemmer

    def build_analyzer(self):
        analyzer = super(StemmedCountVectorizer, self).build_analyzer()
        return lambda doc: (self.stemmer.stem(w) for w in analyzer(doc))

    def get_params(self, deep=True):
        params = super().get_params(deep=deep)
        cp = copy.copy(self)
        cp.__class__ = CountVectorizer
        params.update(CountVectorizer.get_params(cp, deep))
        return params

In [2]:
# -*- coding: utf-8 -*-

"""FakeNewsClassifier Trainer

This script conduct the following steps (overview):

1. Load the processes dataset from {PROJECT_DIR}/data/processed as dataframe
2. Concatenate title and text in one column
3. Split the dataset into train and test

4. Set up a sklearn pipeline with preprocessor and classifier for the combined column (title+text)
   and perform hyperparam search 
5. Extract best score and best_params for and log it
6. Predict on testset and log the score

7. Pickle best pipe to disk (final model)

"""
# %%
import logging
logging.basicConfig(level=logging.INFO)
logger = logging.getLogger()

import os
import pandas as pd
from sklearn.pipeline import Pipeline
from sklearn.compose import ColumnTransformer
from sklearn.ensemble import VotingClassifier
from catboost import CatBoostClassifier, Pool
from sklearn.model_selection import train_test_split, GridSearchCV
from sklearn.metrics import classification_report, confusion_matrix, f1_score, recall_score, precision_score
from dotenv import find_dotenv, load_dotenv
import nltk
from nltk.stem.snowball import GermanStemmer
from sklearn.feature_extraction.text import CountVectorizer
import dagshub
from joblib import dump


load_dotenv(find_dotenv())
#%% Load and split dataset
# INPUTFILE = os.path.join(os.getenv('PROJECT_DIR'), 'data', 'processed', 'fake_news_processed.csv')
INPUTFILE = os.path.join('data', 'datasets_merged.csv')
df = pd.read_csv(INPUTFILE, sep=';')
logger.info('Distribution of fake news in entire dataset: \n%s' % df.fake.value_counts(normalize=True))

X = df['title']+' '+df['text']
y = df['fake']

X_train, X_test, y_train, y_test = train_test_split(X, y,
                    test_size=0.2,
                    stratify=y,
                    random_state=42)

# Download and init stopwordlist
nltk.download('stopwords')
STOPWORD_LIST = nltk.corpus.stopwords.words('german')

#print(X_train[100])

# %% Test with CountVectorizer
"""
count_vec = StemmedCountVectorizer(token_pattern=r'\b[a-zA-Z]{2,}\b',
                                max_features=100,
                                stop_words=STOPWORD_LIST,
                                lowercase=True,
                                stemmer=GermanStemmer(ignore_stopwords=True),
                                ngram_range=(2,2)
                        )

word_counts = count_vec.fit_transform(X_train).todense()

df_body = pd.DataFrame(word_counts, columns=count_vec.get_feature_names())
df_body_trans = pd.DataFrame(df_body.T.sum(axis=1), columns=['count'])
df_body_trans.sort_values(by='count', ascending=False).head(20)

"""

#%% Define param grid for tuning
param_grid = {'vectorizer__max_features':[500],
        'clf__n_estimators': [400],
        'clf__learning_rate': [0.9],
        'clf__max_depth': [5]}

# Constants for training
# MODEL_PATH = os.getenv('MODEL_PATH')
MODEL_PATH = 'model.pkl'
CV_SCORING = 'f1'
CV_FOLDS = 3

# %%
with dagshub.dagshub_logger() as dagslog:
        # Pipeline
        pipe = Pipeline(steps=[
        ('vectorizer', StemmedCountVectorizer(stop_words=STOPWORD_LIST,
                                   token_pattern=r'\b[a-zA-Z]{2,}\b',
                                   stemmer=GermanStemmer(ignore_stopwords=True),
                                   lowercase=True)),
        ('clf', CatBoostClassifier(allow_writing_files=False))
        ])
        
        # pipe_title_tune.get_params().keys()
        logger.info('Start training estimator pipeline')
        cv_grid = GridSearchCV(pipe, param_grid, scoring=CV_SCORING, cv=CV_FOLDS, n_jobs=2)
        cv_grid.fit(X_train, y_train)
        #cv_grid.best_estimator_.named_steps['clf'].get_all_params()

        # Log and assign best score and params to var
        logger.info('Best params from GridSearchCV: %s' % str(cv_grid.best_params_))
        dagslog.log_hyperparams({'model_path': MODEL_PATH})
        dagslog.log_hyperparams({'cv_score': CV_SCORING})
        dagslog.log_hyperparams({'best_cv_params': cv_grid.best_params_})

        logger.info('Best %s from GridSearchCV: %s' % (CV_SCORING, str(cv_grid.best_score_)))
        dagslog.log_metrics({'best_cv_score': cv_grid.best_score_})
        
        # Predict on testdata
        y_pred = cv_grid.best_estimator_.predict(X_test)
        logger.info('f1_score for testdata: %s' % str(f1_score(y_test, y_pred)))
        dagslog.log_metrics({'f1_score_on_testdata': f1_score(y_test, y_pred)})
        logger.info('precision_score for testdata: %s' % str(precision_score(y_test, y_pred)))
        dagslog.log_metrics({'precision_score_on_testdata': precision_score(y_test, y_pred)})
        logger.info('recall_score for testdata: %s' % str(recall_score(y_test, y_pred)))
        dagslog.log_metrics({'recall_score_on_testdata': recall_score(y_test, y_pred)})
        logger.info('Classification report for testdata: \n%s' % classification_report(y_test, y_pred))
        logger.info('Confusion matrix for testdata: \n%s' % confusion_matrix(y_test, y_pred))

        # Dump model to disk
        dump(cv_grid.best_estimator_, MODEL_PATH)

INFO:numexpr.utils:NumExpr defaulting to 4 threads.
INFO:root:Distribution of fake news in entire dataset: 
0    0.922928
1    0.077072
Name: fake, dtype: float64
[nltk_data] Downloading package stopwords to
[nltk_data]     C:\Users\fjun\AppData\Roaming\nltk_data...
[nltk_data]   Unzipping corpora\stopwords.zip.
INFO:root:Start training estimator pipeline


0:	learn: 0.1790348	total: 211ms	remaining: 1m 24s
1:	learn: 0.1292450	total: 269ms	remaining: 53.6s
2:	learn: 0.1143598	total: 327ms	remaining: 43.2s
3:	learn: 0.1028553	total: 376ms	remaining: 37.2s
4:	learn: 0.0949345	total: 428ms	remaining: 33.8s
5:	learn: 0.0891566	total: 472ms	remaining: 31s
6:	learn: 0.0860460	total: 518ms	remaining: 29.1s
7:	learn: 0.0828053	total: 560ms	remaining: 27.4s
8:	learn: 0.0804890	total: 615ms	remaining: 26.7s
9:	learn: 0.0802442	total: 657ms	remaining: 25.6s
10:	learn: 0.0773813	total: 714ms	remaining: 25.2s
11:	learn: 0.0743588	total: 769ms	remaining: 24.9s
12:	learn: 0.0742056	total: 803ms	remaining: 23.9s
13:	learn: 0.0712867	total: 859ms	remaining: 23.7s
14:	learn: 0.0710548	total: 894ms	remaining: 22.9s
15:	learn: 0.0693764	total: 936ms	remaining: 22.5s
16:	learn: 0.0692207	total: 970ms	remaining: 21.9s
17:	learn: 0.0672085	total: 1.02s	remaining: 21.7s
18:	learn: 0.0662156	total: 1.06s	remaining: 21.3s
19:	learn: 0.0660709	total: 1.1s	remaining

165:	learn: 0.0280339	total: 9.32s	remaining: 13.1s
166:	learn: 0.0279555	total: 9.36s	remaining: 13.1s
167:	learn: 0.0278377	total: 9.4s	remaining: 13s
168:	learn: 0.0278304	total: 9.44s	remaining: 12.9s
169:	learn: 0.0278144	total: 9.47s	remaining: 12.8s
170:	learn: 0.0277465	total: 9.51s	remaining: 12.7s
171:	learn: 0.0277411	total: 9.55s	remaining: 12.7s
172:	learn: 0.0276537	total: 9.59s	remaining: 12.6s
173:	learn: 0.0276439	total: 9.63s	remaining: 12.5s
174:	learn: 0.0272863	total: 9.68s	remaining: 12.4s
175:	learn: 0.0272144	total: 9.72s	remaining: 12.4s
176:	learn: 0.0272099	total: 9.76s	remaining: 12.3s
177:	learn: 0.0270794	total: 9.8s	remaining: 12.2s
178:	learn: 0.0270130	total: 9.84s	remaining: 12.1s
179:	learn: 0.0270002	total: 9.88s	remaining: 12.1s
180:	learn: 0.0269826	total: 9.94s	remaining: 12s
181:	learn: 0.0269773	total: 10.1s	remaining: 12s
182:	learn: 0.0268656	total: 10.2s	remaining: 12.1s
183:	learn: 0.0266789	total: 10.3s	remaining: 12.1s
184:	learn: 0.026674

325:	learn: 0.0175428	total: 20.5s	remaining: 4.64s
326:	learn: 0.0174896	total: 20.5s	remaining: 4.58s
327:	learn: 0.0174893	total: 20.5s	remaining: 4.51s
328:	learn: 0.0174874	total: 20.6s	remaining: 4.44s
329:	learn: 0.0174652	total: 20.6s	remaining: 4.37s
330:	learn: 0.0174231	total: 20.7s	remaining: 4.3s
331:	learn: 0.0173994	total: 20.7s	remaining: 4.24s
332:	learn: 0.0173747	total: 20.7s	remaining: 4.17s
333:	learn: 0.0173731	total: 20.8s	remaining: 4.1s
334:	learn: 0.0172833	total: 20.8s	remaining: 4.04s
335:	learn: 0.0172530	total: 20.9s	remaining: 3.97s
336:	learn: 0.0170780	total: 20.9s	remaining: 3.91s
337:	learn: 0.0170560	total: 20.9s	remaining: 3.84s
338:	learn: 0.0170001	total: 21s	remaining: 3.77s
339:	learn: 0.0169427	total: 21s	remaining: 3.71s
340:	learn: 0.0168846	total: 21.1s	remaining: 3.65s
341:	learn: 0.0168833	total: 21.1s	remaining: 3.58s
342:	learn: 0.0168808	total: 21.1s	remaining: 3.51s
343:	learn: 0.0168686	total: 21.2s	remaining: 3.45s
344:	learn: 0.0168

INFO:root:Best params from GridSearchCV: {'clf__learning_rate': 0.9, 'clf__max_depth': 5, 'clf__n_estimators': 400, 'vectorizer__max_features': 500}
INFO:root:Best f1 from GridSearchCV: 0.8555504998353983


399:	learn: 0.0146894	total: 24.9s	remaining: 0us


INFO:root:f1_score for testdata: 0.8681434599156118
INFO:root:precision_score for testdata: 0.8878101402373247
INFO:root:recall_score for testdata: 0.849329205366357
INFO:root:Classification report for testdata: 
              precision    recall  f1-score   support

           0       0.99      0.99      0.99     11599
           1       0.89      0.85      0.87       969

    accuracy                           0.98     12568
   macro avg       0.94      0.92      0.93     12568
weighted avg       0.98      0.98      0.98     12568

INFO:root:Confusion matrix for testdata: 
[[11495   104]
 [  146   823]]


In [9]:
key = 'HOME'
value = os.getenv(key)
  
# Print the value of 'HOME'
# environment variable
print("Value of 'HOME' environment variable :", value) 
  
# Get the value of 'JAVA_HOME'
# environment variable
key = 'MODEL_PATH'
value = os.getenv(key)
  
# Print the value of 'JAVA_HOME'
# environment variable
print("Value of 'JAVA_HOME' environment variable :", value) 

Value of 'HOME' environment variable : None
Value of 'JAVA_HOME' environment variable : None
