In [1]:
from google.colab import drive
drive.mount('/content/drive')

Mounted at /content/drive


In [2]:
import os, sys
sys.path.append('/content/drive/MyDrive/CPD_BT')

In [3]:
import numpy as np
import matplotlib.pyplot as plt
import scipy.stats as stats
import random

import itertools

from bt_cpd import *

import time
import bisect

import pandas as pd

import statsmodels.api as sm
from sklearn import linear_model

from warnings import simplefilter
from sklearn.exceptions import ConvergenceWarning
simplefilter("ignore", category=ConvergenceWarning)

  import pandas.util.testing as tm


In [4]:
T = 4
Delta = 800
m = np.array([Delta] * T)
cp_truth = np.cumsum(m)[:T-1]
print(cp_truth)

n = 20

sub = 0.5

[ 800 1600 2400]


In [5]:
path = '/content/drive/MyDrive/CPD_BT/experiment_random/'
with open(path + 'data_n' + str(n) + '_Delta' + str(Delta) + '_K' + str(T - 1) + '_sub' + str(int(100 * sub)) + '.npy', 'rb') as f:
    beta_list = np.load(f)
    X_train_list = np.load(f)
    Y_train_list = np.load(f)
    X_test_list = np.load(f)
    Y_test_list = np.load(f)

In [6]:
X_train_list.shape

(100, 3200, 20)

In [7]:
np.random.seed(0)

m_intervals = 50
grid_n = 200
gamma_list = [20, 40]
lam_list = [0.1]

nt = Delta * T
B = 100

run_time_wbs = np.zeros(B)
loc_error_wbs = np.zeros(B)
K_wbs = np.zeros(B)

cp_best_list = []
param_best_list = []

for b in range(B):
    X_train = X_train_list[b]
    Y_train = Y_train_list[b]
    X_test = X_test_list[b]
    Y_test = Y_test_list[b]

    start_time = time.time()
    wbs_fit = wbs_cv_bt(m_intervals, grid_n, lam_list, gamma_list, smooth = 5, buffer = 5)
    res = wbs_fit.fit((X_train, Y_train), (X_test, Y_test))
    cp_best, cp_val, cusum_val, threshold_best, grid = res   
    run_time_wbs[b] = time.time() - start_time
    loc_error_wbs[b] = cp_distance(cp_best, cp_truth)
    K_wbs[b] = len(cp_best)

    cp_best_list.append(cp_best)
    param_best_list.append(threshold_best)

    print(b)

print('---------- wbs -----------')
print("avg loc error: {0}, avg time: {1}".format(loc_error_wbs.mean(), run_time_wbs.mean()))
print("std loc error: {0}, std time: {1}".format(loc_error_wbs.std(), run_time_wbs.std()))
print('K < K*: {0}, K = K*: {1}, K > K*: {2}'.format(sum(K_wbs < T - 1), sum(K_wbs == T - 1), sum(K_wbs > T - 1)))

0
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
---------- wbs -----------
avg loc error: 407.52, avg time: 137.2441203761101
std loc error: 336.79490732491786, std time: 21.65641279747018
K < K*: 10, K = K*: 21, K > K*: 69


In [8]:
loc_error_wbs

array([ 368.,  416.,  208.,  256.,  416.,   16.,  560.,  352., 1632.,
         16.,  368.,  224.,   80.,    0.,  272.,  672.,  368.,  336.,
         32.,  224.,  208.,  400.,  576.,  208.,  400.,  272.,  304.,
        448.,  272.,  592.,  560.,  256.,  816.,  384.,  272., 1600.,
        448.,  368.,   32.,  400., 1568.,  400.,   48.,  320.,  352.,
        416.,  784.,  336.,  688.,  272.,  256.,  224.,  224.,  496.,
        784.,  352.,   96.,  800.,  416.,  256.,   80.,  240.,  336.,
        288.,  656.,  608.,  352.,   32., 1600.,  368.,  624.,   16.,
        560.,  224.,  400.,   80.,  256.,    0.,  240.,  336.,  544.,
        640.,  608.,  128.,  560.,   48.,  336.,  416.,  144.,  400.,
         16.,  784.,   16.,  496.,  304.,  464.,  368.,  544., 1552.,
        368.])

In [9]:
cp_best_list

[[432, 592, 832, 1584, 2384],
 [384, 592, 800, 1840, 2288, 2656],
 [800, 1616, 1808, 2400, 2560],
 [800, 1600, 2464, 2656],
 [800, 1504, 2416, 2816],
 [800, 1616, 2400],
 [768, 1600, 2384, 2960],
 [816, 1600, 2048],
 [2432],
 [784, 1600, 2400],
 [624, 800, 1232, 1584, 2400, 2640],
 [640, 800, 1024, 1600, 1808, 2400],
 [720, 1664, 2384],
 [800, 1600, 2400],
 [784, 1552, 2400, 2672],
 [800, 1600, 2432, 3072],
 [800, 1040, 1584, 1968, 2400],
 [640, 816, 1136, 1312, 2416],
 [800, 1568, 2416],
 [752, 1504, 1824, 2400],
 [800, 1008, 1632, 2368],
 [784, 1520, 2432, 2800],
 [224, 448, 848, 1312, 1520, 2384],
 [816, 1008, 1584, 2192, 2416],
 [800, 1648, 1968, 2336, 2560, 2800],
 [800, 1600, 1808, 2128],
 [784, 1104, 1584, 2400],
 [960, 1632, 2384, 2848],
 [816, 1072, 1360, 1600, 2384, 2608],
 [208, 816, 1568, 2480, 2864],
 [800, 1344, 1712, 2064, 2400, 2960],
 [800, 1600, 2400, 2656],
 [1616, 2496],
 [416, 832, 1280, 1664, 2400, 2768],
 [864, 1392, 1600, 1840, 2128, 2416],
 [2400],
 [352, 704, 

In [10]:
import pickle
with open(path + 'wbs_n' + str(n) + '_Delta' + str(Delta) + '_K' + str(T - 1) + '_grid' + str(grid_n) + '_sub' + str(int(100 * sub)) + '.pickle', 'wb') as f:
    pickle.dump([beta_list, cp_best_list, param_best_list, loc_error_wbs, run_time_wbs, K_wbs], f)

In [11]:
loc_error_wbs[loc_error_wbs > (T // 2) * Delta] = (T // 2) * Delta

print("avg loc error: {0}, avg time: {1}".format(loc_error_wbs.mean(), run_time_wbs.mean()))
print("std loc error: {0}, std time: {1}".format(loc_error_wbs.std(), run_time_wbs.std()))
print('K < K*: {0}, K = K*: {1}, K > K*: {2}'.format(sum(K_wbs < T - 1), sum(K_wbs == T - 1), sum(K_wbs > T - 1)))

avg loc error: 407.2, avg time: 137.2441203761101
std loc error: 335.6445739171125, std time: 21.65641279747018
K < K*: 10, K = K*: 21, K > K*: 69
