In [12]:
from sklearn.naive_bayes import MultinomialNB # ideal für counting features wie bow oder tfidf https://towardsdatascience.com/why-how-to-use-the-naive-bayes-algorithms-in-a-regulated-industry-with-sklearn-python-code-dbd8304ab2cf
from sklearn.naive_bayes import GaussianNB # für Features in Decimal Form geeignet
from sklearn.naive_bayes import ComplementNB # ähnlich wie Multinomial, soll sich aber besser für imbalanced data eignen
from sklearn.metrics import (
    accuracy_score,
    confusion_matrix,
    ConfusionMatrixDisplay,
    f1_score,
    recall_score,
    precision_score,
    classification_report,
)
import pandas as pd
from sklearn.model_selection import GridSearchCV, train_test_split, cross_val_score

In [13]:
def evaluate(y_test,y_pred):
    accuracy = accuracy_score(y_test, y_pred)
    f1 = f1_score(y_test, y_pred)
    recall = recall_score(y_test, y_pred)
    precision = precision_score(y_test, y_pred)
    print("Accuracy:", accuracy)
    print("F1 Score:", f1)
    print("Recall:", recall)
    print("Precision:", precision)
    print(pd.DataFrame(confusion_matrix(y_test, y_pred)))

## Evaluation neue Vectorize-Funktionen (08.12.)

In [14]:
%run ../../functions/vectorize_functions.py

In [15]:
filepath_name = (('../../../data/new_datasets/train_cleaned.csv'))
df_cleaned = pd.read_csv(filepath_name, encoding='utf-8')

## GaussianNB

In [16]:
#df_param = [df_cleaned]
#text_column_param = ['tweet_cleaned']
#label_column_param = ['label']
vector_size_param = [10, 50, 100, 200, 300]
window_param = [3, 5, 10, 20, 35]
min_count_param = [1, 3, 5, 10]
#test_size_param = [0.3]
#random_state_param = [42]

In [17]:
from itertools import product
#all_combinations = product(df_param, text_column_param, label_column_param, vector_size_param, window_param, min_count_param, test_size_param, 
 #                          random_state_param)
all_combinations = product(vector_size_param, window_param, min_count_param)

In [18]:
def used_parameters(vector_size=300, window=5, min_count=1):
    vector_size_res = vector_size
    window_res = window
    min_count_res = min_count

    return  vector_size_res, window_res, min_count_res

In [19]:
results_list = []

In [21]:

for combination in all_combinations:
    #X_train, X_test, y_train, y_test = vectorize_w2v(*combination)
    vector_size_used, window_used, min_count_used = used_parameters(*combination)
    X_train, X_test, y_train, y_test = vectorize_w2v(df=df_cleaned, text_column='tweet_cleaned', 
                                                                                 label_column="label", vector_size=vector_size_used,
                                                    window=window_used, min_count=min_count_used)
    
    model = GaussianNB()  
    model.fit(X_train, y_train)  

    y_train_pred = model.predict(X_train)

    y_test_pred = model.predict(X_test)

    train_accuracy = accuracy_score(y_train, y_train_pred)
    train_recall = recall_score(y_train, y_train_pred)
    train_precision = precision_score(y_train, y_train_pred)
    train_f1 = f1_score(y_train, y_train_pred)

    test_accuracy = accuracy_score(y_test, y_test_pred)
    test_recall = recall_score(y_test, y_test_pred)
    test_precision = precision_score(y_test, y_test_pred)
    test_f1 = f1_score(y_test, y_test_pred)

    result_dict = {
        'model': 'W2V Param',
        'vector_size': vector_size_used, 
        'window': window_used,
        'min_count': min_count_used,
        'train_accuracy': train_accuracy,
        'train_recall': train_recall,
        'train_precision': train_precision,
        'train_f1': train_f1,
        'test_accuracy': test_accuracy,
        'test_recall': test_recall,
        'test_precision': test_precision,
        'test_f1': test_f1
    }

    results_list.append(result_dict)

results_df = pd.DataFrame(results_list)

results_df.to_csv('eval_data/nb_grid_w2v_parameters_new_dataset.csv', index=False)

print(results_df)


        model  vector_size  window  min_count  train_accuracy  train_recall  \
0   W2V Param           10       3          3        0.779239      0.152983   
1   W2V Param           10       3          5        0.788125      0.080793   
2   W2V Param           10       3         10        0.786562      0.098323   
3   W2V Param           10       5          1        0.782388      0.135453   
4   W2V Param           10       5          3        0.782504      0.159734   
..        ...          ...     ...        ...             ...           ...   
94  W2V Param          300      20         10        0.705520      0.596472   
95  W2V Param          300      35          1        0.719933      0.566420   
96  W2V Param          300      35          3        0.719280      0.569904   
97  W2V Param          300      35          5        0.716108      0.579922   
98  W2V Param          300      35         10        0.702139      0.591355   

    train_precision  train_f1  test_accuracy  test_