In [67]:
import pandas as pd
import numpy as np
from datetime import datetime

from sklearn.model_selection import train_test_split, GridSearchCV, StratifiedKFold
from sklearn.preprocessing import StandardScaler
from sklearn.feature_selection import SelectKBest,f_classif
from sklearn.metrics import accuracy_score, precision_score,recall_score,f1_score
from sklearn.utils import shuffle

In [87]:
df = pd.read_csv('data/secondary-prediction/secondary_prediction_players_statistics.csv')
shuffle(df.groupby(['game']).size())

game
ESPORTSTMNT02_2578008    2
ESPORTSTMNT02_2574753    2
ESPORTSTMNT02_2557101    2
ESPORTSTMNT05_2520809    2
ESPORTSTMNT01_2706561    2
                        ..
ESPORTSTMNT02_2552458    2
ESPORTSTMNT01_2693060    2
ESPORTSTMNT05_2520335    2
ESPORTSTMNT01_2702196    1
ESPORTSTMNT02_2554335    2
Length: 1328, dtype: int64

In [69]:
y = df['firstTower'].copy()
X = df.drop(['game', 'firstBlood', 'kills', 'deaths', 'firstTower', 'firstHerald', 'dragons', 'barons', 'inhibitors', 'towers', 'heralds'], axis=1)
X

Unnamed: 0,flagSide,topGP,topWR,topKDA,jungleGP,jungleWR,jungleKDA,midGP,midWR,midKDA,carryGP,carryWR,carryKDA,suppGP,suppWR,suppKDA
0,1,3,0.67,4.0,0,0.00,0.0,12,0.42,4.2,1,0.0,4.0,1,1.00,9.0
1,0,6,0.50,5.1,4,0.50,4.7,0,0.00,0.0,1,0.0,1.3,11,0.36,2.4
2,1,4,0.25,2.2,2,0.00,1.2,3,0.33,4.1,0,0.0,0.0,0,0.00,0.0
3,0,4,0.00,1.5,3,0.33,1.1,2,0.50,4.2,4,0.5,3.0,0,0.00,0.0
4,1,3,0.00,1.6,0,0.00,0.0,1,0.00,0.7,5,0.4,3.1,0,0.00,0.0
...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...
2567,1,10,0.60,3.1,1,0.00,1.7,0,0.00,0.0,0,0.0,0.0,3,0.00,1.6
2568,0,0,0.00,0.0,11,0.45,4.3,1,0.00,0.0,2,0.0,0.8,0,0.00,0.0
2569,1,0,0.00,0.0,0,0.00,0.0,0,0.00,0.0,2,0.5,7.0,1,0.00,1.7
2570,0,1,0.00,2.4,1,1.00,2.6,0,0.00,0.0,0,0.0,0.0,3,0.67,2.9


In [70]:
def preprocess_input(X,y):
    X = X.copy()
    X_train,X_test,y_train,y_test = train_test_split(X,y,train_size=0.8,random_state=42, stratify=y)
    scaler = StandardScaler()   
    print(X_train) 
    scaler.fit(X_train)
    X_train = scaler.transform(X_train)
    X_test = scaler.transform(X_test)
    return X_train,X_test,y_train,y_test
X_train,X_test,y_train,y_test = preprocess_input(X,y)

      flagSide  topGP  topWR  topKDA  jungleGP  jungleWR  jungleKDA  midGP  \
1680         0      7   0.29     1.7         6      0.67        5.2      0   
567          1      1   1.00     3.4         2      1.00        4.6      3   
1958         1     13   0.46     2.0         1      1.00        3.4      3   
1686         0      2   0.50     1.4        25      0.64        4.7      0   
1314         0     11   0.73     5.4        10      0.60       11.3      8   
...        ...    ...    ...     ...       ...       ...        ...    ...   
1080         1      0   0.00     0.0         3      0.67       10.0      5   
1002         1      0   0.00     0.0         8      0.38        2.6      2   
470          1      1   1.00     7.0         2      0.00        1.8      6   
980          1      0   0.00     0.0         1      1.00        7.5      0   
2449         1      0   0.00     0.0         9      0.67        5.5      7   

      midWR  midKDA  carryGP  carryWR  carryKDA  suppGP  suppWR

In [71]:
from sklearn.model_selection import train_test_split
from sklearn.model_selection import GridSearchCV
from sklearn.preprocessing import StandardScaler

from sklearn.linear_model import LogisticRegression
from sklearn.svm import LinearSVC, SVC
from sklearn.tree import DecisionTreeClassifier
from sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier, AdaBoostClassifier

models = {
    'Logistic Regression': LogisticRegression(max_iter=1000),
    'Support Vector Machine (Linear Kernel)': LinearSVC(),
    'Support Vector Machine (RBF Kernel)': SVC(),
    'Decission Tree': DecisionTreeClassifier(),
    'Adaboost': AdaBoostClassifier(),
    'Random Forest': RandomForestClassifier(),
    'Gradient Boosting Classifier': GradientBoostingClassifier()
}

GBCmodel = GradientBoostingClassifier()
GBCmodel.fit(X_train,y_train)


for name, model in models.items():
  model.fit(X_train,y_train)
  print(name + ' trained')

Logistic Regression trained
Support Vector Machine (Linear Kernel) trained




Support Vector Machine (RBF Kernel) trained
Decission Tree trained
Adaboost trained
Random Forest trained
Gradient Boosting Classifier trained


In [72]:
scores_list = []

for name,model in models.items():    
    scores_list.append({
    'Model': name,
    'Accuracy': accuracy_score(y_test,model.predict(X_test)),
    'Precision':  precision_score(y_test,model.predict(X_test)),
    'Recall': recall_score(y_test,model.predict(X_test)),
    'F1-Score': f1_score(y_test,model.predict(X_test))
    })
scores = pd.DataFrame(scores_list)

In [73]:
scores

Unnamed: 0,Model,Accuracy,Precision,Recall,F1-Score
0,Logistic Regression,0.526214,0.526502,0.57529,0.549815
1,Support Vector Machine (Linear Kernel),0.528155,0.528169,0.579151,0.552486
2,Support Vector Machine (RBF Kernel),0.533981,0.534296,0.571429,0.552239
3,Decission Tree,0.52233,0.526316,0.501931,0.513834
4,Adaboost,0.500971,0.503846,0.505792,0.504817
5,Random Forest,0.518447,0.520295,0.544402,0.532075
6,Gradient Boosting Classifier,0.528155,0.529851,0.548263,0.538899
