In [2]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.neighbors import KNeighborsClassifier
from sklearn.preprocessing import LabelEncoder, StandardScaler
import matplotlib.pyplot as plt

df = pd.read_csv('../rnb_features.csv')
df = df.dropna()  # drop rows with missing values

# Encode categorical features
encoder = LabelEncoder()
df['key'] = encoder.fit_transform(df['key'])
df['scale'] = encoder.fit_transform(df['scale'])
df['genre'] = encoder.fit_transform(df['genre'])  # this is the label

df['genre_binary'] = df['genre'].apply(lambda x: 1 if x.lower() == 'rnb' else 0) #converts the genre

features = [
    'bpm',
    'danceability',
    'average_loudness',
    'spectral_centroid',
    'chords_strength',
    'onset_rate',
    'key',
    'scale'
]
# Features and labels
X = df[features]
y = df['genre']

# scale features
scaler = StandardScaler()
X_scaled = scaler.fit_transform(X)

# Train/test split
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2)

# Train kNN
model = KNeighborsClassifier(n_neighbors=7)
model.fit(X_train, y_train)

print("Accuracy:", model.score(X_test, y_test))

Accuracy: 0.03046898337150254


In [None]:
# Confusion matrix
from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay
import matplotlib.pyplot as plt

y_pred = model.predict(X_test)
cm = confusion_matrix(y_test, y_pred)
disp = ConfusionMatrixDisplay(confusion_matrix=cm)
disp.plot(xticks_rotation=45)
plt.title('Confusion Matrix')
plt.show()

In [None]:
from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay
import matplotlib.pyplot as plt

y_pred = model.predict(X_test)
cm = confusion_matrix(y_test, y_pred)
cm_df = pd.DataFrame(cm, index=['Not R&B', 'R&B'], columns=['Not R&B', 'R&B'])
plt.figure(figsize=(10,7))
sns.heatmap(cm_df, annot=True, fmt='d', cmap='Blues')
plt.title('Confusion Matrix')
plt.xlabel('Predicted')
plt.ylabel('Actual')
plt.show()