In [26]:
import pandas as pd
from sklearn.linear_model import LinearRegression
from sklearn.ensemble import RandomForestRegressor
from sklearn.neural_network import MLPRegressor
from sklearn.svm import SVR
from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_squared_error, r2_score
from sklearn.impute import SimpleImputer
from sklearn.tree import DecisionTreeRegressor
from sklearn.svm import SVR
from sklearn.pipeline import make_pipeline
from sklearn.compose import make_column_transformer
from sklearn.preprocessing import StandardScaler
from sklearn.ensemble import GradientBoostingRegressor

# See all columns with head, a personal preference
pd.set_option('display.max_columns', 500)

In [27]:
df = pd.read_csv('./data/df_clean.csv')

In [28]:
# Linear Regression
X = df.drop('Interest Rate', axis=1)
y = df['Interest Rate']

X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
imputer = SimpleImputer(strategy='mean')
X_train = imputer.fit_transform(X_train)
X_test = imputer.transform(X_test)

model = LinearRegression()
model.fit(X_train, y_train)

y_train_pred = model.predict(X_train)
y_test_pred = model.predict(X_test)

r2_train = r2_score(y_train, y_train_pred)
r2_test = r2_score(y_test, y_test_pred)

print("R2 score for training set:", r2_train)
print("R2 score for testing set:", r2_test)


R2 score for training set: 0.923440714580506
R2 score for testing set: 0.9258464733527175


In [29]:
# Decision Tree

X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
imputer = SimpleImputer(strategy='mean')
X_train = imputer.fit_transform(X_train)
X_test = imputer.transform(X_test)

model = DecisionTreeRegressor(random_state=42)
model.fit(X_train, y_train)

y_train_pred = model.predict(X_train)
y_test_pred = model.predict(X_test)

r2_train = r2_score(y_train, y_train_pred)
r2_test = r2_score(y_test, y_test_pred)

print("Decision Tree Regressor: R2 score for training set:", r2_train)
print("Decision Tree Regressor: R2 score for testing set:", r2_test)

Decision Tree Regressor: R2 score for training set: 0.9995677460648379
Decision Tree Regressor: R2 score for testing set: 0.9961602781586563


In [30]:
# Gradient Boosting

X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)
imputer = SimpleImputer(strategy='mean')
X_train = imputer.fit_transform(X_train)
X_test = imputer.transform(X_test)

model = GradientBoostingRegressor(random_state=42)
model.fit(X_train, y_train)

y_train_pred = model.predict(X_train)
y_test_pred = model.predict(X_test)

r2_train = r2_score(y_train, y_train_pred)
r2_test = r2_score(y_test, y_test_pred)

print("Gradient Boosting Regressor: R2 score for training set:", r2_train)
print("Gradient Boosting Regressor: R2 score for testing set:", r2_test)

Gradient Boosting Regressor: R2 score for training set: 0.914363605470526
Gradient Boosting Regressor: R2 score for testing set: 0.9171763825657623
