In [13]:
import sys
sys.path.append("../")
import numpy as np
import time
import pandas as pd
import pickle
import math
from typing import Tuple


from sklearn.feature_extraction.text import TfidfVectorizer
from sklearn.svm import SVR
from sklearn.linear_model import LinearRegression
from sklearn.ensemble import RandomForestRegressor
from lightgbm import LGBMRegressor
from catboost import CatBoostRegressor
from sklearn.metrics import mean_squared_error, mean_absolute_error

from scipy.sparse import hstack

ANALYSIS_POSTFIX = "mined_sudden_2024-08-26"

experiment_config = {
    "RS" : 42,
    "ANALYSIS_POSTFIX": ANALYSIS_POSTFIX,
    "FEATURE_MODE" : "CODE_MODEL", # CODE_MODEL
}

In [14]:
def step_two(experiment_config, 
             X_train,
             y_train,
             model,
             X_val=None,
             y_val=None,
             save=False): 
    
    ANALYSIS_POSTFIX = experiment_config["ANALYSIS_POSTFIX"]
    
    training_start_time = time.time()
    if model=="lr":
        reg = LinearRegression().fit(X_train, y_train)
    elif model =="svm": 
        reg = SVR().fit(X_train, y_train)
    elif model=="rf":
        reg = RandomForestRegressor.fit(X_train, y_train)
    elif model=="lgbm":
        reg = LGBMRegressor(max_depth=10, silent=True)
        reg.fit(X=X_train, y=y_train)
    elif model=="catboost":
        reg = CatBoostRegressor()
        reg.fit(X=X_train, y=y_train)
    training_end_time = time.time()
    time_training = training_end_time - training_start_time

    
    if save:
        with open(f'./models/reg_{model}_{ANALYSIS_POSTFIX}.pkl','wb') as f:
            pickle.dump(reg, f)
        return f'./models/reg_{model}_{ANALYSIS_POSTFIX}.pkl'
    
    else:
        inference_start_time = time.time()
        y_pred = reg.predict(X_val)
        inference_end_time = time.time()
        time_inference = inference_end_time - inference_start_time

        y_pred[y_pred<0] = 0
        mae = mean_absolute_error(y_true=y_val, y_pred=y_pred)
        rmse = math.sqrt(mean_squared_error(y_true=y_val, y_pred=y_pred))
        return {"pred": y_pred, "mae": mae, "rmse": rmse, "time_training" : time_training, "time_inference" : time_inference}
    

def cv_step_2(experiment_config:dict, cv_df:pd.DataFrame) -> Tuple:

    t_models = ["lr", "svm", "lgbm", "catboost"]

    results = {}

    FEATURE_MODE = experiment_config["FEATURE_MODE"]

    for test_fold in range(cv_df.fold.max()+1):
        print(test_fold)

        # Prepare the input data
        vectorizer = TfidfVectorizer()
        X_train_tfidf = vectorizer.fit_transform(cv_df.loc[cv_df.fold!=test_fold, "input_sequence"])

        if FEATURE_MODE=="CODE_MODEL":
            X_train_column_sparse = pd.get_dummies(cv_df.loc[cv_df.fold!=test_fold, "model_set"], sparse=True).sparse.to_coo().tocsr()
            X_train = hstack([X_train_column_sparse, X_train_tfidf])
        elif FEATURE_MODE=="CODE":
            X_train = X_train_tfidf
            
        y_train = cv_df.loc[cv_df.fold!=test_fold, "rouge"]
        
        X_val_tfidf = vectorizer.transform(cv_df.loc[cv_df.fold==test_fold, "input_sequence"])
        if FEATURE_MODE=="CODE_MODEL":
            X_val_column_sparse = pd.get_dummies(cv_df.loc[cv_df.fold==test_fold, "model_set"], sparse=True).sparse.to_coo().tocsr()
            X_val = hstack([X_val_column_sparse, X_val_tfidf])
        elif FEATURE_MODE=="CODE":
            X_val = X_val_tfidf
            
        y_val = cv_df.loc[cv_df.fold==test_fold, "rouge"]

        results[test_fold] = {}
        for model in t_models:
            print(model)
            preds_df = step_two(experiment_config=experiment_config,
                                X_train=X_train,
                                y_train=y_train,
                                X_val=X_val,
                                y_val=y_val,
                                model=model)
            cv_df.loc[cv_df.fold==test_fold, f"{model}_perf_hat"] = preds_df["pred"]
            results[test_fold][model] = preds_df

    cv_df = cv_df.reset_index(drop=True)

    return cv_df

def full_step_2(cv_df:pd.DataFrame,
                experiment_config:dict) -> None:
    
    ANALYSIS_POSTFIX = experiment_config["ANALYSIS_POSTFIX"]
    # TRAIN ON ALL PREDICTIONS AT ONCE

    t_models = ["lr", "svm", "lgbm", "catboost"]
    FEATURE_MODE = experiment_config["FEATURE_MODE"]

    # Prepare the input data
    vectorizer = TfidfVectorizer()
    X_train_tfidf = vectorizer.fit_transform(cv_df.loc[cv_df.model_set!="ensemble", "input_sequence"])
    if FEATURE_MODE=="CODE_MODEL":
        X_train_column_sparse = pd.get_dummies(cv_df.loc[cv_df.model_set!="ensemble", "model_set"], sparse=True).sparse.to_coo().tocsr()
        X_train = hstack([X_train_column_sparse, X_train_tfidf])
    elif FEATURE_MODE=="CODE":
        X_train = X_train_tfidf
        
    y_train = cv_df.loc[cv_df.model_set!="ensemble", "rouge"]
        
    with open(f"./models/vectorizer_{ANALYSIS_POSTFIX}.pkl", "wb") as file:
        pickle.dump(vectorizer, file, protocol=pickle.HIGHEST_PROTOCOL) 
        
    for model in t_models:
        print(model)
        preds_df = step_two(experiment_config=experiment_config,
                            X_train=X_train,
                            y_train=y_train,
                            model=model,
                            save=True)
        
def pred_perf(experiment_config,
              X,
              model): 

    ANALYSIS_POSTFIX = experiment_config["ANALYSIS_POSTFIX"]

    with open(f'./models/reg_{model}_{ANALYSIS_POSTFIX}.pkl','rb') as f:
            reg = pickle.load(f)

    y_pred = reg.predict(X)
    y_pred[y_pred<0] = 0
    return y_pred

def meta_predict(experiment_config:dict, 
                 test_df: pd.DataFrame,
                 base_models_names: list,
                 t_models:list = ["svm", "catboost"]) -> pd.DataFrame:

    ANALYSIS_POSTFIX = experiment_config["ANALYSIS_POSTFIX"]
    FEATURE_MODE = experiment_config["FEATURE_MODE"]
    
    for model_i, model_set in enumerate(base_models_names):

        set_df = test_df.copy()
        set_df["model_set"] = model_set
        # Prepare the input data
        with open(f"./models/vectorizer_{ANALYSIS_POSTFIX}.pkl", "rb") as file:
            vectorizer = pickle.load(file)

        if model_i==0:
            meta_preds_df = set_df.copy()
        else: 
            meta_preds_df = pd.concat([meta_preds_df, set_df])
            
    X_test_tfidf = vectorizer.transform(meta_preds_df.loc[:, "input_sequence"])
    if FEATURE_MODE=="CODE_MODEL":
        X_test_column_sparse = pd.get_dummies(meta_preds_df.loc[:, "model_set"], sparse=True).sparse.to_coo().tocsr()
        X_test = hstack([X_test_column_sparse, X_test_tfidf])
    elif FEATURE_MODE=="CODE":
        X_test = X_test_tfidf

    for model in t_models:
        print(model)
        meta_preds_df[f"{model}_preds"] = pred_perf(experiment_config=experiment_config, 
                                                    X=X_test,
                                                    model=model)

    meta_preds_df = meta_preds_df.reset_index(drop=True)
    return meta_preds_df

In [15]:
with open(f"../ensemble_learning/reports/results/{ANALYSIS_POSTFIX}/cv_results.pickle", "rb") as handle:
    cv_predictions = pickle.load(handle)

with open(f"../ensemble_learning/reports/results/{ANALYSIS_POSTFIX}/test_results.pickle", "rb") as handle:
    test_predictions = pickle.load(handle)


In [16]:
COLUMNS_TEST = ['question_id', 'parent_answer_post_id', 'prob', 'input_sequence',
       'output_sequence', 'id', 'snippet_len', 'intent_len', 'snippet_token_n',
       'intent_token_n', 'cluster', 'input_ids', 'attention_mask', 'labels',
       'prediction', 'rouge', 'model_set']

COLUMNS_CV = COLUMNS_TEST.copy()
COLUMNS_CV.append("fold")

#### Preprocessing

In [17]:
cv_predictions = cv_predictions.loc[cv_predictions.model_set!="ensemble", COLUMNS_CV]
test_predictions = test_predictions.loc[cv_predictions.model_set!="ensemble", COLUMNS_TEST]

# Code Only

We have 9 base lerner settings models that we compare learning of 1, splitting to two meta models,  all together. 

In [18]:
MODELS_LIST = [0, 1, 2, 5, 10, 'cluster_[1]', 'cluster_[4]', 'cluster_[3]', 'cluster_[0, 1, 4]']
MODE = ["ONE-BY-ONE", "TWO-MODELS", "ALL"]

In [19]:
results_cv_df = pd.DataFrame()

t_models = ["lr", "svm", "lgbm", "catboost"]

temp_df =  cv_predictions.copy()
temp_df = cv_step_2(experiment_config=experiment_config,
            cv_df=temp_df)

for model_meta in t_models:
    for cluster in sorted(temp_df.cluster.unique()):

        print(cluster)
        cluster_temp_df = temp_df.loc[temp_df.cluster==cluster, :]


        mae = mean_absolute_error(y_true=cluster_temp_df.loc[:, "rouge"],
                                    y_pred=cluster_temp_df.loc[:, f"{model_meta}_perf_hat"])
        
        rmse = math.sqrt(mean_squared_error(y_true=cluster_temp_df.loc[:, "rouge"],
                                    y_pred=cluster_temp_df.loc[:, f"{model_meta}_perf_hat"]))
        
        t_res = pd.DataFrame(data={"model_base": "all", "model_meta": model_meta, "cluster": cluster, "rmse": rmse, "mae": mae}, index=[0])


        results_cv_df = pd.concat([results_cv_df, t_res], axis=0)
    


for model_meta in t_models:


    mae = mean_absolute_error(y_true=temp_df.loc[:, "rouge"],
                                    y_pred=temp_df.loc[:, f"{model_meta}_perf_hat"])
    
    rmse = math.sqrt(mean_squared_error(y_true=temp_df.loc[:, "rouge"],
                                    y_pred=temp_df.loc[:, f"{model_meta}_perf_hat"]))
    
    t_res = pd.DataFrame(data={"model_base": "all", "model_meta": model_meta, "cluster": "full", "rmse": rmse, "mae": mae,}, index=[0])


    results_cv_df = pd.concat([results_cv_df, t_res], axis=0)

results_cv_df = results_cv_df.sort_values(["model_meta", "cluster"])


0
lr
svm
lgbm
[LightGBM] [Info] Auto-choosing row-wise multi-threading, the overhead of testing was 2.212944 seconds.
You can set `force_row_wise=true` to remove the overhead.
And if memory is not enough, you can set `force_col_wise=true`.
[LightGBM] [Info] Total Bins 13269
[LightGBM] [Info] Number of data points in the train set: 41994, number of used features: 1041
[LightGBM] [Info] Start training from score 0.229969
catboost
Learning rate set to 0.0739
0:	learn: 0.1673837	total: 9.67ms	remaining: 9.66s
1:	learn: 0.1659756	total: 16.8ms	remaining: 8.37s
2:	learn: 0.1647401	total: 24.7ms	remaining: 8.22s
3:	learn: 0.1637148	total: 31.9ms	remaining: 7.94s
4:	learn: 0.1628099	total: 38.8ms	remaining: 7.73s
5:	learn: 0.1620342	total: 45.4ms	remaining: 7.52s
6:	learn: 0.1613456	total: 52ms	remaining: 7.38s
7:	learn: 0.1607092	total: 59.5ms	remaining: 7.38s
8:	learn: 0.1601290	total: 66.4ms	remaining: 7.31s
9:	learn: 0.1596523	total: 73.1ms	remaining: 7.24s
10:	learn: 0.1591993	total: 80ms

In [None]:
print("MAE")
display(results_cv_df.groupby(["model_meta", "cluster"], as_index=False)["mae"].describe())

MAE


Unnamed: 0,model_meta,cluster,count,mean,std,min,25%,50%,75%,max
0,catboost,0,1.0,0.132278,,0.132278,0.132278,0.132278,0.132278,0.132278
1,catboost,1,1.0,0.140277,,0.140277,0.140277,0.140277,0.140277,0.140277
2,catboost,2,1.0,0.135758,,0.135758,0.135758,0.135758,0.135758,0.135758
3,catboost,3,1.0,0.138211,,0.138211,0.138211,0.138211,0.138211,0.138211
4,catboost,4,1.0,0.140148,,0.140148,0.140148,0.140148,0.140148,0.140148
5,catboost,full,1.0,0.136847,,0.136847,0.136847,0.136847,0.136847,0.136847
6,lgbm,0,1.0,0.131779,,0.131779,0.131779,0.131779,0.131779,0.131779
7,lgbm,1,1.0,0.137186,,0.137186,0.137186,0.137186,0.137186,0.137186
8,lgbm,2,1.0,0.135394,,0.135394,0.135394,0.135394,0.135394,0.135394
9,lgbm,3,1.0,0.137212,,0.137212,0.137212,0.137212,0.137212,0.137212


In [None]:
print("RMSE")
display(results_cv_df.groupby(["model_meta", "cluster"], as_index=False)["rmse"].describe())

RMSE


Unnamed: 0,model_meta,cluster,count,mean,std,min,25%,50%,75%,max
0,catboost,0,1.0,0.164342,,0.164342,0.164342,0.164342,0.164342,0.164342
1,catboost,1,1.0,0.175061,,0.175061,0.175061,0.175061,0.175061,0.175061
2,catboost,2,1.0,0.167085,,0.167085,0.167085,0.167085,0.167085,0.167085
3,catboost,3,1.0,0.171641,,0.171641,0.171641,0.171641,0.171641,0.171641
4,catboost,4,1.0,0.171309,,0.171309,0.171309,0.171309,0.171309,0.171309
5,catboost,full,1.0,0.169001,,0.169001,0.169001,0.169001,0.169001,0.169001
6,lgbm,0,1.0,0.162507,,0.162507,0.162507,0.162507,0.162507,0.162507
7,lgbm,1,1.0,0.170927,,0.170927,0.170927,0.170927,0.170927,0.170927
8,lgbm,2,1.0,0.166132,,0.166132,0.166132,0.166132,0.166132,0.166132
9,lgbm,3,1.0,0.169855,,0.169855,0.169855,0.169855,0.169855,0.169855


In [None]:
results_test_df = pd.DataFrame()

t_models = ["lr", "svm", "lgbm", "catboost"]

cv_temp_df =  cv_predictions.copy()
temp_df =  test_predictions.copy()
full_step_2(experiment_config=experiment_config,
                        cv_df=cv_temp_df)
temp_df = meta_predict(experiment_config=experiment_config,
                        test_df=temp_df,
                        base_models_names=MODELS_LIST,
                        t_models=t_models)

for model_meta in t_models:
    for cluster in sorted(temp_df.cluster.unique()):

        print(cluster)
        cluster_temp_df = temp_df.loc[temp_df.cluster==cluster, :]


        mae = mean_absolute_error(y_true=cluster_temp_df.loc[:, "rouge"],
                                    y_pred=cluster_temp_df.loc[:, f"{model_meta}_preds"])
        
        rmse = math.sqrt(mean_squared_error(y_true=cluster_temp_df.loc[:, "rouge"],
                                    y_pred=cluster_temp_df.loc[:, f"{model_meta}_preds"]))
        
        t_res = pd.DataFrame(data={"model_base": "all", "model_meta": model_meta, "cluster": cluster, "rmse": rmse, "mae": mae}, index=[0])

        results_test_df = pd.concat([results_test_df, t_res], axis=0)
    
for model_meta in t_models:


    mae = mean_absolute_error(y_true=temp_df.loc[:, "rouge"],
                                    y_pred=temp_df.loc[:, f"{model_meta}_preds"])
    
    rmse = math.sqrt(mean_squared_error(y_true=temp_df.loc[:, "rouge"],
                                    y_pred=temp_df.loc[:, f"{model_meta}_preds"]))
    
    t_res = pd.DataFrame(data={"model_base": "all", "model_meta": model_meta, "cluster": "full", "rmse": rmse, "mae": mae,}, index=[0])

    results_test_df = pd.concat([results_test_df, t_res], axis=0)


results_test_df = results_test_df.sort_values(["model_meta", "cluster"])


lr
svm
lgbm
[LightGBM] [Info] Auto-choosing row-wise multi-threading, the overhead of testing was 9.968824 seconds.
You can set `force_row_wise=true` to remove the overhead.
And if memory is not enough, you can set `force_col_wise=true`.
[LightGBM] [Info] Total Bins 19304
[LightGBM] [Info] Number of data points in the train set: 63000, number of used features: 1353
[LightGBM] [Info] Start training from score 0.229955
catboost
Learning rate set to 0.078791
0:	learn: 0.1687946	total: 10.3ms	remaining: 10.3s
1:	learn: 0.1685870	total: 18.4ms	remaining: 9.2s
2:	learn: 0.1683899	total: 26.4ms	remaining: 8.77s
3:	learn: 0.1682254	total: 34.6ms	remaining: 8.61s
4:	learn: 0.1680447	total: 42.6ms	remaining: 8.47s
5:	learn: 0.1678856	total: 50.6ms	remaining: 8.39s
6:	learn: 0.1677471	total: 58.6ms	remaining: 8.32s
7:	learn: 0.1676141	total: 66.7ms	remaining: 8.27s
8:	learn: 0.1675029	total: 74.8ms	remaining: 8.23s
9:	learn: 0.1673853	total: 83ms	remaining: 8.22s
10:	learn: 0.1672789	total: 91.1m

In [None]:
print("MAE")
display(results_test_df.groupby(["model_meta", "cluster"], as_index=False)["mae"].describe())

MAE


Unnamed: 0,model_meta,cluster,count,mean,std,min,25%,50%,75%,max
0,catboost,0,1.0,0.118195,,0.118195,0.118195,0.118195,0.118195,0.118195
1,catboost,1,1.0,0.122414,,0.122414,0.122414,0.122414,0.122414,0.122414
2,catboost,2,1.0,0.129886,,0.129886,0.129886,0.129886,0.129886,0.129886
3,catboost,3,1.0,0.13334,,0.13334,0.13334,0.13334,0.13334,0.13334
4,catboost,4,1.0,0.133558,,0.133558,0.133558,0.133558,0.133558,0.133558
5,catboost,full,1.0,0.133055,,0.133055,0.133055,0.133055,0.133055,0.133055
6,lgbm,0,1.0,0.116169,,0.116169,0.116169,0.116169,0.116169,0.116169
7,lgbm,1,1.0,0.11888,,0.11888,0.11888,0.11888,0.11888,0.11888
8,lgbm,2,1.0,0.130665,,0.130665,0.130665,0.130665,0.130665,0.130665
9,lgbm,3,1.0,0.134996,,0.134996,0.134996,0.134996,0.134996,0.134996


In [None]:
print("RMSE")
display(results_test_df.groupby(["model_meta", "cluster"], as_index=False)["rmse"].describe())

RMSE


Unnamed: 0,model_meta,cluster,count,mean,std,min,25%,50%,75%,max
0,catboost,0,1.0,0.141522,,0.141522,0.141522,0.141522,0.141522,0.141522
1,catboost,1,1.0,0.152188,,0.152188,0.152188,0.152188,0.152188,0.152188
2,catboost,2,1.0,0.15952,,0.15952,0.15952,0.15952,0.15952,0.15952
3,catboost,3,1.0,0.1653,,0.1653,0.1653,0.1653,0.1653,0.1653
4,catboost,4,1.0,0.163578,,0.163578,0.163578,0.163578,0.163578,0.163578
5,catboost,full,1.0,0.163105,,0.163105,0.163105,0.163105,0.163105,0.163105
6,lgbm,0,1.0,0.139602,,0.139602,0.139602,0.139602,0.139602,0.139602
7,lgbm,1,1.0,0.144875,,0.144875,0.144875,0.144875,0.144875,0.144875
8,lgbm,2,1.0,0.160026,,0.160026,0.160026,0.160026,0.160026,0.160026
9,lgbm,3,1.0,0.166718,,0.166718,0.166718,0.166718,0.166718,0.166718
