In [1]:
import pandas as pd
import numpy as np
from sklearn.ensemble import RandomForestClassifier
from sklearn.grid_search import GridSearchCV
from sklearn import cross_validation, metrics

import matplotlib.pylab as plt
%matplotlib inline



In [2]:
train = pd.read_csv('./mlpractice/data/train_modified.csv')
target='Disbursed' # Disbursed的值就是二元分类的输出
IDcol = 'ID'
train['Disbursed'].value_counts()

0    19680
1      320
Name: Disbursed, dtype: int64

In [3]:
x_columns = [x for x in train.columns if x not in [target, IDcol]]
X = train[x_columns]
y = train['Disbursed']

In [4]:
rf0 = RandomForestClassifier(oob_score=True, random_state=10)
rf0.fit(X,y)
print (rf0.oob_score_)
y_predprob = rf0.predict_proba(X)[:,1]
print ("AUC Score (Train): %f" % metrics.roc_auc_score(y,
                                            y_predprob))

0.98005
AUC Score (Train): 0.999833


  warn("Some inputs do not have OOB scores. "
  predictions[k].sum(axis=1)[:, np.newaxis])


In [5]:
param_test1 = {'n_estimators':[10,71,10]}
gsearch1 = GridSearchCV(estimator = RandomForestClassifier(min_samples_split=100,
                                  min_samples_leaf=20,max_depth=8,max_features='sqrt' ,random_state=10), 
                       param_grid = param_test1, scoring='roc_auc',cv=5)
gsearch1.fit(X,y)
gsearch1.grid_scores_, gsearch1.best_params_, gsearch1.best_score_

([mean: 0.80681, std: 0.02236, params: {'n_estimators': 10},
  mean: 0.81945, std: 0.02840, params: {'n_estimators': 71},
  mean: 0.80681, std: 0.02236, params: {'n_estimators': 10}],
 {'n_estimators': 71},
 0.8194546335111786)

In [8]:
param_test2 = {'max_depth':[3,14,2], 'min_samples_split':[50,201,20]}
gsearch2 = GridSearchCV(estimator = RandomForestClassifier(n_estimators= 60, 
                                  min_samples_leaf=20,max_features='sqrt' ,oob_score=True, random_state=10),
   param_grid = param_test2, scoring='roc_auc',iid=False, cv=5)
gsearch2.fit(X,y)
gsearch2.grid_scores_, gsearch2.best_params_, gsearch2.best_score_

([mean: 0.79379, std: 0.02347, params: {'max_depth': 3, 'min_samples_split': 50},
  mean: 0.79349, std: 0.02542, params: {'max_depth': 3, 'min_samples_split': 201},
  mean: 0.79377, std: 0.02406, params: {'max_depth': 3, 'min_samples_split': 20},
  mean: 0.82226, std: 0.02116, params: {'max_depth': 14, 'min_samples_split': 50},
  mean: 0.81859, std: 0.02711, params: {'max_depth': 14, 'min_samples_split': 201},
  mean: 0.82085, std: 0.02291, params: {'max_depth': 14, 'min_samples_split': 20},
  mean: 0.78546, std: 0.02676, params: {'max_depth': 2, 'min_samples_split': 50},
  mean: 0.78549, std: 0.02669, params: {'max_depth': 2, 'min_samples_split': 201},
  mean: 0.78546, std: 0.02676, params: {'max_depth': 2, 'min_samples_split': 20}],
 {'max_depth': 14, 'min_samples_split': 50},
 0.8222604643038618)

In [8]:
rf1 = RandomForestClassifier(n_estimators= 60, 
max_depth=13, min_samples_split=110,
min_samples_leaf=20,max_features='sqrt',
oob_score=True, random_state=10)
rf1.fit(X,y)
print (rf1.oob_score_)

0.984


In [11]:
param_test3 = {'min_samples_split':[80,150,20],
               'min_samples_leaf': [10,60,1]}
gsearch3 = GridSearchCV(estimator = RandomForestClassifier(n_estimators= 60, 
   max_depth=13, max_features='sqrt' ,oob_score=True, random_state=10),
   param_grid = param_test3, scoring='roc_auc',iid=False, cv=5)
gsearch3.fit(X,y)
gsearch3.grid_scores_, gsearch3.best_params_, gsearch3.best_score_

([mean: 0.82093, std: 0.02287, params: {'min_samples_leaf': 10, 'min_samples_split': 80},
  mean: 0.81683, std: 0.02337, params: {'min_samples_leaf': 10, 'min_samples_split': 150},
  mean: 0.81954, std: 0.01710, params: {'min_samples_leaf': 10, 'min_samples_split': 20},
  mean: 0.81766, std: 0.02457, params: {'min_samples_leaf': 60, 'min_samples_split': 80},
  mean: 0.81625, std: 0.02442, params: {'min_samples_leaf': 60, 'min_samples_split': 150},
  mean: 0.81766, std: 0.02457, params: {'min_samples_leaf': 60, 'min_samples_split': 20},
  mean: 0.81989, std: 0.02576, params: {'min_samples_leaf': 1, 'min_samples_split': 80},
  mean: 0.81485, std: 0.02634, params: {'min_samples_leaf': 1, 'min_samples_split': 150},
  mean: 0.81246, std: 0.02095, params: {'min_samples_leaf': 1, 'min_samples_split': 20}],
 {'min_samples_leaf': 10, 'min_samples_split': 80},
 0.8209294016768294)

In [16]:
param_test4 = {'max_features':[3,11,2]}
gsearch4 = GridSearchCV(estimator =
            RandomForestClassifier(
            n_estimators= 60, max_depth=13, 
            min_samples_split=120,
            min_samples_leaf=20 ,
            oob_score=True, random_state=10),
            param_grid = param_test4, 
            scoring='roc_auc',iid=False, cv=5)
gsearch4.fit(X,y)
gsearch4.grid_scores_, gsearch4.best_params_, gsearch4.best_score_

([mean: 0.81981, std: 0.02586, params: {'max_features': 3},
  mean: 0.81897, std: 0.02200, params: {'max_features': 11},
  mean: 0.81258, std: 0.03527, params: {'max_features': 2}],
 {'max_features': 3},
 0.8198119124745936)

In [16]:
rf2 = RandomForestClassifier(n_estimators= 60, max_depth=13, min_samples_split=120,
                                  min_samples_leaf=20,max_features=7 ,oob_score=True, random_state=10)
rf2.fit(X,y)
print(rf2.oob_score_)

0.984
