In [22]:
from sklearn.datasets import load_iris
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.neighbors import KNeighborsClassifier
from sklearn.model_selection import GridSearchCV
from sklearn.tree import DecisionTreeClassifier, export_graphviz


In [23]:
def knn_iris():
    """
    用 KNN 算法对鸢尾花进行分类
    """
    # 1. 获取数据
    iris = load_iris()

    # 2. 划分数据集
    x_train, x_test, y_train, y_test = train_test_split(
        iris.data, iris.target, random_state=6)

    # 3. 特征工程: 标准化
    transfer = StandardScaler()
    x_train = transfer.fit_transform(x_train)
    x_test = transfer.transform(x_test)

    # 4. KNN 算法预估器
    estimator = KNeighborsClassifier(n_neighbors=3)
    estimator.fit(x_train, y_train)

    # 5. 模型评估
    # 方法1: 直接比对真实值和预测值
    y_predict = estimator.predict(x_test)
    print("y_predict: \n", y_predict)
    print("直接比对真实值和预测值: \n", y_test == y_predict)

    # 方法2: 计算准确率
    score = estimator.score(x_test, y_test)
    print("准确率为: \n", score)
    return None


In [24]:
def knn_iris_gscv():
    """
    用 KNN 算法对鸢尾花进行分类, 添加网络搜索和交叉验证
    """
    # 1) 获取数据
    iris = load_iris()

    # 2) 划分数据集
    x_train, x_test, y_train, y_test = train_test_split(
        iris.data, iris.target, random_state=22)

    # 3) 特征工程: 标准化
    transfer = StandardScaler()
    x_train = transfer.fit_transform(x_train)
    x_test = transfer.transform(x_test)

    # 4) KNN 算法预估器
    estimator = KNeighborsClassifier()

    # 加入网络搜索与交叉验证
    # 参数准备
    param_dict = {"n_neighbors": [1, 3, 5, 7, 9, 11]}
    estimator = GridSearchCV(estimator, param_grid=param_dict, cv=10)
    estimator.fit(x_train, y_train)

    # 5) 模型评估
    # 方法1: 直接对比真实值和预测值
    y_predict = estimator.predict(x_test)
    print("y_predict: \n", y_predict)
    print("直接对比真实值和预测值: \n", y_test == y_predict)

    # 方法2: 计算准确率
    score = estimator.score(x_test, y_test)
    print("准确率为: \n", score)

    print("最佳参数为: \n", estimator.best_params_)
    print("最佳结果为: \n", estimator.best_score_)
    print("最佳估计器为: \n", estimator.best_estimator_)
    print("最终交叉验证结果为: \n", estimator.cv_results_)

    return None


In [25]:
def decision_iris():
    """
    用决策树对鸢尾花进行分类
    """
    # 1) 获取数据集
    iris = load_iris()

    # 2) 划分数据集
    x_train, x_test, y_train, y_test = train_test_split(
        iris.data, iris.target, random_state=22)

    # 3) 决策树预估器
    estimator = DecisionTreeClassifier(criterion="entropy")
    estimator.fit(x_train, y_train)

    # 4) 模型评估
    # 方法1: 直接对比真实值和预测值
    y_predict = estimator.predict(x_test)
    print("y_predict: \n", y_predict)
    print("直接对比真实值和预测值: \n", y_test == y_predict)

    # 方法2: 计算准确率
    score = estimator.score(x_test, y_test)
    print("准确率为: \n", score)

    # 可视化决策树
    export_graphviz(estimator, out_file="iris_tree.dot")

    return None


In [26]:
if __name__ == "__main__":
    # code1: 用 KNN 算法对鸢尾花进行分类
    # knn_iris()
    # code2: 利用KNN算法对鸢尾花进行分类, 添加网格搜索和交叉验证
    # knn_iris_gscv()

    # code4: 利用决策树对鸢尾花进行分类
    decision_iris()

y_predict: 
 [0 2 1 2 1 1 1 1 1 0 2 1 2 2 0 2 1 1 1 1 0 2 0 1 2 0 2 2 2 1 0 0 1 1 1 0 0
 0]
直接对比真实值和预测值: 
 [ True  True  True  True  True  True  True False  True  True  True  True
  True  True  True  True  True  True False  True  True  True  True  True
  True  True  True  True  True False  True  True  True  True  True  True
  True  True]
准确率为: 
 0.9210526315789473
