In [70]:
import pandas as pd
import numpy as np

# Reading match data

In [20]:
matches = pd.read_csv("matches.csv", index_col=0)

In [21]:
matches.head()

Unnamed: 0,date,time,comp,round,day,venue,result,gf,ga,opponent,...,match report,notes,sh,sot,dist,fk,pk,pkatt,season,team
1,2021-08-15,16:30,Premier League,Matchweek 1,Sun,Away,L,0.0,1.0,Tottenham,...,Match Report,,18.0,4.0,16.9,1.0,0.0,0.0,2022,Manchester City
2,2021-08-21,15:00,Premier League,Matchweek 2,Sat,Home,W,5.0,0.0,Norwich City,...,Match Report,,16.0,4.0,17.3,1.0,0.0,0.0,2022,Manchester City
3,2021-08-28,12:30,Premier League,Matchweek 3,Sat,Home,W,5.0,0.0,Arsenal,...,Match Report,,25.0,10.0,14.3,0.0,0.0,0.0,2022,Manchester City
4,2021-09-11,15:00,Premier League,Matchweek 4,Sat,Away,W,1.0,0.0,Leicester City,...,Match Report,,25.0,8.0,14.0,0.0,0.0,0.0,2022,Manchester City
6,2021-09-18,15:00,Premier League,Matchweek 5,Sat,Home,D,0.0,0.0,Southampton,...,Match Report,,16.0,1.0,15.7,1.0,0.0,0.0,2022,Manchester City


In [22]:
matches.shape

(1389, 27)

# Investigating missing data

In [23]:
38*20*2 #38 matches, 20 teams, 2 seasons

1520

In [24]:
matches["team"].value_counts()

team
Southampton                 72
Brighton and Hove Albion    72
Manchester United           72
West Ham United             72
Newcastle United            72
Burnley                     71
Leeds United                71
Crystal Palace              71
Manchester City             71
Wolverhampton Wanderers     71
Tottenham Hotspur           71
Arsenal                     71
Leicester City              70
Chelsea                     70
Aston Villa                 70
Everton                     70
Liverpool                   38
Fulham                      38
West Bromwich Albion        38
Sheffield United            38
Brentford                   34
Watford                     33
Norwich City                33
Name: count, dtype: int64

In [25]:
matches["round"].value_counts()

round
Matchweek 1     39
Matchweek 16    39
Matchweek 34    39
Matchweek 32    39
Matchweek 31    39
Matchweek 29    39
Matchweek 28    39
Matchweek 26    39
Matchweek 25    39
Matchweek 24    39
Matchweek 23    39
Matchweek 2     39
Matchweek 19    39
Matchweek 17    39
Matchweek 20    39
Matchweek 15    39
Matchweek 5     39
Matchweek 3     39
Matchweek 13    39
Matchweek 12    39
Matchweek 4     39
Matchweek 11    39
Matchweek 10    39
Matchweek 9     39
Matchweek 8     39
Matchweek 14    39
Matchweek 7     39
Matchweek 6     39
Matchweek 30    37
Matchweek 27    37
Matchweek 22    37
Matchweek 21    37
Matchweek 18    37
Matchweek 33    32
Matchweek 35    20
Matchweek 36    20
Matchweek 37    20
Matchweek 38    20
Name: count, dtype: int64

In [26]:
matches["date"] = pd.to_datetime(matches["date"])

# Creating predictors

In [27]:
matches["venue_code"] = matches["venue"].astype("category").cat.codes #string -> category -> num

In [28]:
matches["opp_code"] = matches["opponent"].astype("category").cat.codes #string -> category -> num

In [29]:
matches["hour"] = matches["time"].str.replace(":.+","",regex = True).astype("int")

In [32]:
matches["day_code"] = matches["date"].dt.dayofweek

In [34]:
matches["target"] = (matches["result"] == "W").astype("int")

In [35]:
matches

Unnamed: 0,date,time,comp,round,day,venue,result,gf,ga,opponent,...,fk,pk,pkatt,season,team,venue_code,opp_code,hour,day_code,target
1,2021-08-15,16:30,Premier League,Matchweek 1,Sun,Away,L,0.0,1.0,Tottenham,...,1.0,0.0,0.0,2022,Manchester City,0,18,16,6,0
2,2021-08-21,15:00,Premier League,Matchweek 2,Sat,Home,W,5.0,0.0,Norwich City,...,1.0,0.0,0.0,2022,Manchester City,1,15,15,5,1
3,2021-08-28,12:30,Premier League,Matchweek 3,Sat,Home,W,5.0,0.0,Arsenal,...,0.0,0.0,0.0,2022,Manchester City,1,0,12,5,1
4,2021-09-11,15:00,Premier League,Matchweek 4,Sat,Away,W,1.0,0.0,Leicester City,...,0.0,0.0,0.0,2022,Manchester City,0,10,15,5,1
6,2021-09-18,15:00,Premier League,Matchweek 5,Sat,Home,D,0.0,0.0,Southampton,...,1.0,0.0,0.0,2022,Manchester City,1,17,15,5,0
...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...,...
38,2021-05-02,19:15,Premier League,Matchweek 34,Sun,Away,L,0.0,4.0,Tottenham,...,0.0,0.0,0.0,2021,Sheffield United,0,18,19,6,0
39,2021-05-08,15:00,Premier League,Matchweek 35,Sat,Home,L,0.0,2.0,Crystal Palace,...,1.0,0.0,0.0,2021,Sheffield United,1,6,15,5,0
40,2021-05-16,19:00,Premier League,Matchweek 36,Sun,Away,W,1.0,0.0,Everton,...,0.0,0.0,0.0,2021,Sheffield United,0,7,19,6,1
41,2021-05-19,18:00,Premier League,Matchweek 37,Wed,Away,L,0.0,1.0,Newcastle Utd,...,1.0,0.0,0.0,2021,Sheffield United,0,14,18,2,0


# Creating ML models

## Spliting data into training and testing sets

In [97]:
predictors = ["venue_code", "opp_code", "hour", "day_code"]
X = matches.iloc[:, -5:-1].values
y = matches.iloc[:, -1].values

In [47]:
print(X)

[[ 0 18 16  6]
 [ 1 15 15  5]
 [ 1  0 12  5]
 ...
 [ 0  7 19  6]
 [ 0 14 18  2]
 [ 1  4 16  6]]


In [48]:
print(y)

[0 1 1 ... 1 0 1]


In [98]:
train = matches[matches["date"] < '2022-01-01']
test = matches[matches["date"] > '2022-01-01']
X_train = train[predictors]
X_test = test[predictors]
y_train = train["target"]
y_test = test["target"]

In [99]:
print(X_train)

    venue_code  opp_code  hour  day_code
1            0        18    16         6
2            1        15    15         5
3            1         0    12         5
4            0        10    15         5
6            1        17    15         5
..         ...       ...   ...       ...
38           0        18    19         6
39           1         6    15         5
40           0         7    19         6
41           0        14    18         2
42           1         4    16         6

[1107 rows x 4 columns]


In [100]:
print(X_test)

    venue_code  opp_code  hour  day_code
31           1         5    12         5
32           0        17    17         5
34           1         2    19         2
35           0        15    17         5
37           1        18    17         5
..         ...       ...   ...       ...
33           0         9    14         6
34           0         3    15         5
35           1         4    14         6
36           0        13    15         5
37           1        14    15         5

[276 rows x 4 columns]


In [101]:
print(y_train)

1     0
2     1
3     1
4     1
6     0
     ..
38    0
39    0
40    1
41    0
42    1
Name: target, Length: 1107, dtype: int32


In [102]:
print(y_test)

31    1
32    0
34    1
35    1
37    0
     ..
33    0
34    0
35    1
36    0
37    0
Name: target, Length: 276, dtype: int32


In [103]:
from sklearn.preprocessing import StandardScaler
sc = StandardScaler()
X_train = sc.fit_transform(X_train)
X_test = sc.transform(X_test)

In [104]:
print(X_train)

[[-1.00090375  1.10827847 -0.18843531  0.87792057]
 [ 0.99909707  0.65106197 -0.56026756  0.3698923 ]
 [ 0.99909707 -1.63502051 -1.67576431  0.3698923 ]
 ...
 [-1.00090375 -0.56818202  0.92706144  0.87792057]
 [-1.00090375  0.49865647  0.55522919 -1.15419249]
 [ 0.99909707 -1.02539851 -0.18843531  0.87792057]]


In [105]:
print(X_test)

[[ 0.99909707 -0.87299301 -1.67576431  0.3698923 ]
 [-1.00090375  0.95587297  0.18339694  0.3698923 ]
 [ 0.99909707 -1.33020951  0.92706144 -1.15419249]
 ...
 [ 0.99909707 -1.02539851 -0.93209981  0.87792057]
 [-1.00090375  0.34625097 -0.56026756  0.3698923 ]
 [ 0.99909707  0.49865647 -0.56026756  0.3698923 ]]


## K-nearest neighbors

In [134]:
from sklearn.neighbors import KNeighborsClassifier
knn = KNeighborsClassifier(n_neighbors = 5, metric = 'minkowski', p = 2)
knn.fit(X_train, y_train)

In [135]:
y_pred = knn.predict(X_test)

In [136]:
accuracy_score(y_test, y_pred)

0.5615942028985508

## Support Vector Machine

In [137]:
from sklearn.svm import SVC
svm = SVC(kernel = 'linear', random_state = 0)
svm.fit(X_train, y_train)

In [138]:
y_pred = svm.predict(X_test)

In [139]:
accuracy_score(y_test, y_pred)

0.6231884057971014

## Random Forest Classifier

In [143]:
from sklearn.ensemble import RandomForestClassifier
rf = RandomForestClassifier(n_estimators = 50, min_samples_split = 10, criterion = 'entropy', random_state = 1)
rf.fit(X_train, y_train)

In [144]:
y_pred = rf.predict(X_test)

In [145]:
accuracy_score(y_test, y_pred)

0.6086956521739131

In [None]:
combined = pd.DataFrame(dict(actual = )