In [1]:
import numpy as np
import pandas as pd
from sklearn.metrics import silhouette_samples, silhouette_score
import matplotlib.cm as cm
import matplotlib.pyplot as plt
import datetime
from importlib import reload

import a2a_clustering
import a2a_validation
import a2a_travellingsalesman
import a2a_kmeans_equalsize

a2a_clustering = reload(a2a_clustering)
a2a_validation = reload(a2a_validation)
a2a_kmeans_equalsize = reload(a2a_kmeans_equalsize)
a2a_travellingsalesman = reload(a2a_travellingsalesman)

n_clusters = 20
RANDOM_SEED = 0

start = datetime.datetime.now()
PATH = 'output/clustering/'
FILE_PREFIX = PATH + 'kmeans_equal_size_' + str(n_clusters) + '_'

df = pd.read_csv("output/data_preparation/first_visit.20190903.csv",
    parse_dates=['created_at'], date_parser=lambda x: pd.datetime.strptime(x, '%Y-%m-%d %H:%M:%S'))

X = a2a_clustering.transform(df)

print("Clustering started")
clusterer = a2a_kmeans_equalsize.EqualGroupsKMeans(n_clusters=n_clusters).fit(X)

df = df.assign(**{
    'Cluster_labels': clusterer.labels_
})

seconds = (datetime.datetime.now() - start).seconds
print("Elapsed time for clustering: " + str(seconds) + " seconds")

centroid_csv = np.asarray(clusterer.cluster_centers_)
np.savetxt(FILE_PREFIX + "centroids.csv", 
    centroid_csv, 
    header="lat,lng", 
    delimiter=",", 
    comments='')

###################
# VALIDATION STEP #
###################

df = a2a_validation.silhouette(df, clusterer.cluster_centers_, FILE_PREFIX, "KMeans")
df.to_csv(FILE_PREFIX + "clusterized_dataset.csv")

###################
# TSP        STEP #
###################

tsp_solved = a2a_travellingsalesman.tsp(df, FILE_PREFIX)
tsp_solved.to_csv(FILE_PREFIX + 'tsp.csv')



Clustering started
Elapsed time for clustering: 2800 seconds
For n_clusters = 20 The average silhouette_score is : 0.2614591881171829


<Figure size 1800x700 with 2 Axes>

In [2]:
tsp_solved

Unnamed: 0,cluster,or_dist,meters,time,time_emptying,seconds,bins,waypoints
0,0,23780.7,23 Km 780.70 m.,0:56:46,1:58:46,3406.8,62.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
1,1,31943.2,31 Km 943.20 m.,1:13:55,2:23:55,4435.5,70.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
2,2,28820.399,28 Km 820.40 m.,1:07:00,2:17:00,4020.3,70.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
3,3,21769.3,21 Km 769.30 m.,0:47:04,1:57:04,2824.2,70.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
4,4,31315.399,31 Km 315.40 m.,1:08:12,2:18:12,4092.1,70.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
5,5,24520.0,24 Km 520.00 m.,0:54:23,2:04:23,3263.6,70.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
6,6,33603.699,33 Km 603.70 m.,1:18:00,2:28:00,4680.9,70.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
7,7,38838.3,38 Km 838.30 m.,1:25:37,2:33:37,5137.1,68.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
8,8,28177.398,28 Km 177.40 m.,1:00:59,2:10:59,3659.7,70.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
9,9,22110.1,22 Km 110.10 m.,0:50:50,2:00:50,3050.0,70.0,"[{""serial"": -1, ""coords"": [45.5069182, 9.26845..."
