In [1]:
import pandas as pd
from sklearn.ensemble import GradientBoostingClassifier
from sklearn.model_selection import GridSearchCV, RandomizedSearchCV, train_test_split
from sklearn.metrics import classification_report, accuracy_score

In [2]:
df = pd.read_csv('heart.csv')
X = df.drop(columns=['target'])
y = df['target']
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=0)

In [3]:
model = GradientBoostingClassifier(random_state=0)
model.fit(X_train, y_train)

predictions = model.predict(X_test)

print(classification_report(y_test, predictions))

              precision    recall  f1-score   support

           0       0.78      0.78      0.78        27
           1       0.82      0.82      0.82        34

    accuracy                           0.80        61
   macro avg       0.80      0.80      0.80        61
weighted avg       0.80      0.80      0.80        61



In [15]:
param_grid = {
    'learning_rate': [0.25, 0.1, 0.05, 0.01],
    'n_estimators': [15, 16, 17, 19],
    'max_depth': [1, 2, 3, 4, 5],
    'min_samples_split': [0.6, 0.7, 0.8, 0.9, 2], #0.7, 0.8, 0.9, 1.0
    'min_samples_leaf': [0.1, 0.2, 0.3, 0.4, 1],
    'max_features': [2, 3, 4]
}

grid = GridSearchCV(model, param_grid, refit=True, verbose=3, n_jobs=-1)
grid.fit(X_train, y_train)
print(grid.best_params_)

grid_pred = grid.predict(X_test)
print(classification_report(y_test, grid_pred))

Fitting 5 folds for each of 6000 candidates, totalling 30000 fits
{'learning_rate': 0.25, 'max_depth': 1, 'max_features': 2, 'min_samples_leaf': 1, 'min_samples_split': 0.6, 'n_estimators': 19}
              precision    recall  f1-score   support

           0       0.85      0.81      0.83        27
           1       0.86      0.88      0.87        34

    accuracy                           0.85        61
   macro avg       0.85      0.85      0.85        61
weighted avg       0.85      0.85      0.85        61

