In [12]:
# Initial imports
import pandas as pd
from sklearn.cluster import KMeans
import plotly.express as px
import hvplot.pandas

In [13]:
# Load data
file_path = "Resources/shopping_data_cleaned.csv"
df_shopping = pd.read_csv(file_path)
df_shopping.head(10)

Unnamed: 0,Card Member,Age,Annual Income,Spending Score (1-100)
0,1,19.0,15000,39.0
1,1,21.0,15000,81.0
2,0,20.0,16000,6.0
3,0,23.0,16000,77.0
4,0,31.0,17000,40.0
5,0,22.0,17000,76.0
6,0,35.0,18000,6.0
7,0,23.0,18000,94.0
8,1,64.0,19000,3.0
9,0,30.0,19000,72.0


In [14]:
inertia = []
k = list(range(1, 11))
# Calculate the inertia for the range of K values
for i in k:
   km = KMeans(n_clusters=i, random_state=0)
   km.fit(df_shopping)
   inertia.append(km.inertia_)

In [15]:
# Create the Elbow Curve using hvPlot
elbow_data = {"k":k, "inertia": inertia}
df_elbow = pd.DataFrame(elbow_data)
df_elbow.hvplot.line(x="k",y="inertia",xticks=k, title="Elbow Curve")

In [16]:
def get_clusters(k, data):
   # Create a copy of the DataFrame
   data = data.copy()

   # Initialize the K-Means model
   model = KMeans(n_clusters=k, random_state=0)

   # Fit the model
   model.fit(data)

   # Predict clusters
   predictions = model.predict(data)

   # Create return DataFrame with predicted clusters
   data["class"] = model.labels_

   return data

In [22]:
five_clusters = get_clusters(5,df_shopping)
five_clusters.head()

Unnamed: 0,Card Member,Age,Annual Income,Spending Score (1-100),class
0,1,19.0,15000,39.0,0
1,1,21.0,15000,81.0,0
2,0,20.0,16000,6.0,0
3,0,23.0,16000,77.0,0
4,0,31.0,17000,40.0,0


In [23]:
six_clusters = get_clusters(6,df_shopping)
six_clusters.head()

Unnamed: 0,Card Member,Age,Annual Income,Spending Score (1-100),class
0,1,19.0,15000,39.0,5
1,1,21.0,15000,81.0,5
2,0,20.0,16000,6.0,5
3,0,23.0,16000,77.0,5
4,0,31.0,17000,40.0,5


In [24]:
# Plotting the 2D-Scatter with x="Annual Income" and y="Spending Score (1-100)"
five_clusters.hvplot.scatter(x="Annual Income", y="Spending Score (1-100)", by="class")

In [25]:
# Plot the 3D-scatter with x="Annual Income", y="Spending Score (1-100)" and z="Age"
fig = px.scatter_3d(
    five_clusters,
    x="Age",
    y="Spending Score (1-100)",
    z="Annual Income",
    color="class",
    symbol="class",
    width=800,
)
fig.update_layout(legend=dict(x=0, y=1))
fig.show()

In [26]:
# Plotting the 3D-Scatter with x="Annual Income", y="Spending Score (1-100)" and z="Age"
fig = px.scatter_3d(
    six_clusters,
    x="Age",
    y="Spending Score (1-100)",
    z="Annual Income",
    color="class",
    symbol="class",
    width=800,
)
fig.update_layout(legend=dict(x=0, y=1))
fig.show()