In [1]:
import numpy as np
import pandas as pd
import datetime

import matplotlib
import matplotlib.pyplot as plt

In [2]:
from sklearn.ensemble import GradientBoostingRegressor
from sklearn.ensemble import RandomForestRegressor
from sklearn.model_selection import train_test_split
from sklearn.model_selection import GridSearchCV
from sklearn.metrics import classification_report
from sklearn.model_selection import validation_curve
from sklearn.model_selection import learning_curve
from sklearn import metrics
from sklearn.metrics import fbeta_score, make_scorer

In [3]:
def plot_learning_curve(estimator, title, X, y, ylim=None, cv=None,
                        n_jobs=1, train_sizes=np.linspace(.1, 1.0, 5)):
    """
    Generate a simple plot of the test and training learning curve.

    Parameters
    ----------
    estimator : object type that implements the "fit" and "predict" methods
        An object of that type which is cloned for each validation.

    title : string
        Title for the chart.

    X : array-like, shape (n_samples, n_features)
        Training vector, where n_samples is the number of samples and
        n_features is the number of features.

    y : array-like, shape (n_samples) or (n_samples, n_features), optional
        Target relative to X for classification or regression;
        None for unsupervised learning.

    ylim : tuple, shape (ymin, ymax), optional
        Defines minimum and maximum yvalues plotted.

    cv : int, cross-validation generator or an iterable, optional
        Determines the cross-validation splitting strategy.
        Possible inputs for cv are:
          - None, to use the default 3-fold cross-validation,
          - integer, to specify the number of folds.
          - An object to be used as a cross-validation generator.
          - An iterable yielding train/test splits.

        For integer/None inputs, if ``y`` is binary or multiclass,
        :class:`StratifiedKFold` used. If the estimator is not a classifier
        or if ``y`` is neither binary nor multiclass, :class:`KFold` is used.

        Refer :ref:`User Guide <cross_validation>` for the various
        cross-validators that can be used here.

    n_jobs : integer, optional
        Number of jobs to run in parallel (default 1).
    """
    plt.figure()
    plt.title(title)
    if ylim is not None:
        plt.ylim(*ylim)
    plt.xlabel("Training examples")
    plt.ylabel("Score")
    train_sizes, train_scores, test_scores = learning_curve(
        estimator, X, y, cv=cv, n_jobs=n_jobs, train_sizes=train_sizes)
    train_scores_mean = np.mean(train_scores, axis=1)
    train_scores_std = np.std(train_scores, axis=1)
    test_scores_mean = np.mean(test_scores, axis=1)
    test_scores_std = np.std(test_scores, axis=1)
    plt.grid()

    plt.fill_between(train_sizes, train_scores_mean - train_scores_std,
                     train_scores_mean + train_scores_std, alpha=0.1,
                     color="r")
    plt.fill_between(train_sizes, test_scores_mean - test_scores_std,
                     test_scores_mean + test_scores_std, alpha=0.1, color="g")
    plt.plot(train_sizes, train_scores_mean, 'o-', color="r",
             label="Training score")
    plt.plot(train_sizes, test_scores_mean, 'o-', color="g",
             label="Cross-validation score")

    plt.legend(loc="best")
    return plt

In [1]:
def Mape(ground_truth, predictions):
    diff = (ground_truth - predictions) / ground_truth
    diff = np.abs(diff)
    mape = diff.sum() / len(ground_truth)
    print('mape=%f'%mape)
    return mape

Mapeloss = make_scorer(Mape, greater_is_better=False)

NameError: name 'make_scorer' is not defined

In [5]:
link_df = pd.read_csv('../data/original/links (table 3).csv', index_col=0)

In [6]:
morning_gbdt_regressor_dict = {}
for link_id in link_df.index.values:

    X_train_df = pd.read_csv('../data/feature/feature2/morning/X_train_link' + str(link_id) + '.csv', index_col = 0)
    y_train_df = pd.read_csv('../data/feature/feature2/morning/y_train_link' + str(link_id) + '.csv', index_col = 0)
    morning_gbdt_regressor_dict[link_id] = []
    for t in range(6):
        X_train, X_validation, y_train, y_validation = train_test_split(X_train_df.values, y_train_df[str(t)].values, random_state=7, test_size=0.5)
        gbdt_regressor = GradientBoostingRegressor(learning_rate=0.01, n_estimators=8,min_samples_split=3, min_samples_leaf=2)
        gbdt_regressor.fit(X_train, y_train)
        predict = gbdt_regressor.predict(X_validation)
        print(gbdt_regressor.score(X_validation, y_validation))
        morning_gbdt_regressor_dict[link_id].append(gbdt_regressor)
        

0.00426265831586
-1.70381019312
-0.00338456304639
-0.00494961510539
0.0107186079191
-0.0483862592131
-0.342201842701
-4.78392446748
-0.469975730697
-0.210613535884
-0.0508376962292
-0.0179691487803
-0.011758522868
-0.491422501821
-0.0176528162215
-0.0802830574605
-0.0329605342964
-0.0978563740807
-0.318475446514
-0.0581138276651
-0.0301665010521
-0.0270865441309
0.029116606276
-0.0116226080327
-0.0111456166072
-0.0211312785028
0.00834365818387
-0.0986043759961
-0.280969161851
-0.00507444189994
0.0055168159737
-1.99924574047
-0.0241063029513
-0.0909404504341
0.00108876483089
-0.0132210087374
-0.0144151639065
-0.0599192197816
-0.0537982288987
-0.0134643437964
-0.0318012102486
-0.0348662778595
-0.00872452030556
-0.0252205681371
0.0117282127028
-0.0314984686046
-0.127277144289
-0.0230334033415
-0.0106209478109
-0.00586291926136
-0.113446489182
-0.0547849102624
-0.247329086348
-0.0027261916072
-0.0420521029932
-0.0160198211982
-0.105060884931
-0.0650271379477
-0.0288830239993
-0.06869232988

In [7]:
afternoon_gbdt_regressor_dict = {}
for link_id in link_df.index.values:

    X_train_df = pd.read_csv('../data/feature/feature2/afternoon/X_train_link' + str(link_id) + '.csv', index_col = 0)
    y_train_df = pd.read_csv('../data/feature/feature2/afternoon/y_train_link' + str(link_id) + '.csv', index_col = 0)
    afternoon_gbdt_regressor_dict[link_id] = []
    for t in range(6):
        X_train, X_validation, y_train, y_validation = train_test_split(X_train_df.values, y_train_df[str(t)].values, random_state=7, test_size=0.5)
        gbdt_regressor = GradientBoostingRegressor(learning_rate=0.01, n_estimators=8,min_samples_split=3, min_samples_leaf=2)
        gbdt_regressor.fit(X_train, y_train)
        predict = gbdt_regressor.predict(X_validation)
        print(gbdt_regressor.score(X_validation, y_validation))
        afternoon_gbdt_regressor_dict[link_id].append(gbdt_regressor)
        

-0.0177486462125
-0.0388131430239
-0.0565609653994
-0.0408922616222
-0.544562131327
-0.015694193931
-0.208105143935
-0.0401010392508
0.0262817020991
-0.0526543392258
-0.275742510767
-0.125608085635
-0.0143825576579
-0.0292089727063
-0.0107271937666
-0.0205918808025
-0.0739956972909
-0.0369993085913
-0.0223510844862
-0.0229833807594
-0.0797757014544
-0.151140780034
-0.0574210192329
-0.0193697925183
-0.0150346061938
0.0191670131866
0.0146235195051
0.00313514552738
-0.055928953468
-0.587399700705
0.00952516685746
-0.0271064685631
0.00214855659225
-0.0221377752628
-0.158925609173
-0.0485788821999
-0.00564907262508
-0.0381188467219
0.00157401506247
-0.00872595020948
-6.07413892616
-30.1443664237
-0.023641062275
-0.0243661223324
-0.228891674635
-0.00506749982802
-9.35923417637
-0.0533572471654
-0.0355224370525
-0.0186570982061
-0.0023812327803
0.0267991478673
-0.00549168262143
-0.0142388264231
-0.00483022162841
-0.0281904599273
0.0251995586276
-0.0575879697948
-0.0846380493309
-3.05741742467

# Test集

In [8]:
morning_predict_link_arrive_time_dict = {}
for link_id in link_df.index.values:

    X_test_df = pd.read_csv('../data/feature/feature2/morning/X_test_link' + str(link_id) + '.csv', index_col = 0)
    morning_predict_link_arrive_time_dict[link_id] = []
    for d in range(len(X_test_df)):
        predicts = []
        for t in range(6):
            gbdt_regressor = morning_gbdt_regressor_dict[link_id][t]
            p = gbdt_regressor.predict(X_test_df.loc[d].values.reshape(1, -1))
            predicts.extend(p)
        morning_predict_link_arrive_time_dict[link_id].append(predicts)

In [9]:
afternoon_predict_link_arrive_time_dict = {}
for link_id in link_df.index.values:

    X_test_df = pd.read_csv('../data/feature/feature2/afternoon/X_test_link' + str(link_id) + '.csv', index_col = 0)
    afternoon_predict_link_arrive_time_dict[link_id] = []
    for d in range(len(X_test_df)):
        predicts = []
        for t in range(6):
            gbdt_regressor = afternoon_gbdt_regressor_dict[link_id][t]
            p = gbdt_regressor.predict(X_test_df.loc[d].values.reshape(1, -1))
            predicts.extend(p)
            
        afternoon_predict_link_arrive_time_dict[link_id].append(predicts)

In [10]:
afternoon_predict_link_arrive_time_dict[122]

[[3.8316093740771655,
  2.6645219512622926,
  5.4038025794487563,
  4.3652533478295599,
  42.367843712593299,
  4.7376597316978479],
 [3.6890442534400738,
  2.6645219512622926,
  5.4038025794487563,
  4.0112555327248822,
  25.675571721857846,
  4.0227614183819727],
 [3.6890442534400738,
  2.6645219512622926,
  5.4038025794487563,
  4.0112555327248822,
  25.675571721857846,
  3.9028524805244724],
 [3.6890442534400738,
  2.4539892889699741,
  5.7810023787453062,
  4.3652533478295599,
  25.580048627032536,
  4.0227614183819727],
 [3.6890442534400738,
  2.4740973270134061,
  5.7810023787453062,
  4.4057693994604206,
  42.393344671944419,
  4.8575686695553486],
 [3.6890442534400738,
  2.6645219512622926,
  5.4038025794487563,
  4.0450349509639008,
  25.578060748577414,
  3.9028524805244724],
 [3.8316093740771655,
  2.6645219512622926,
  5.4038025794487563,
  4.3652533478295599,
  25.58968125188732,
  3.1669002097696928]]

In [11]:
routes = pd.read_csv('../data/original/routes (table 4).csv')

In [12]:
submissions = []
for i in routes.index:
    link_seq = routes.loc[i]['link_seq']
    links = link_seq.split(',')
    
    #morning
    start_time = pd.to_datetime('2016-10-18 08:00:00')
    for d in range(7):
        current_time = start_time
        for t in range(6):
            time = 0
            record = []
            record.append(routes.loc[i]['intersection_id'])
            record.append(routes.loc[i]['tollgate_id'])
            record.append('[' + str(current_time) + ',' + str(current_time + datetime.timedelta(minutes=20)) + ')')
            for link_id in links:
                time = time + morning_predict_link_arrive_time_dict[int(link_id)][d][t]
            record.append(time)

            submissions.append(record)
            current_time = current_time + datetime.timedelta(minutes=20)
            
        start_time = start_time + datetime.timedelta(days=1)
        
    #afternoon
    start_time = pd.to_datetime('2016-10-18 17:00:00')
    for d in range(7):
        current_time = start_time
        for t in range(6):
            time = 0
            record = []
            record.append(routes.loc[i]['intersection_id'])
            record.append(routes.loc[i]['tollgate_id'])
            record.append('[' + str(current_time) + ',' + str(current_time + datetime.timedelta(minutes=20)) + ')')
            for link_id in links:
                time = time + afternoon_predict_link_arrive_time_dict[int(link_id)][d][t]
            record.append(time)

            submissions.append(record)
            current_time = current_time + datetime.timedelta(minutes=20)
            
        start_time = start_time + datetime.timedelta(days=1)

In [13]:
predict_df = pd.DataFrame(submissions).rename(columns={0:'intersection_id', 1:'tollgate_id', 2:'time_window', 3:'avg_travel_time'})

In [14]:
predict_df.to_csv('../data/prediction/prediction_m2.csv', index=False)