In [1]:
import numpy as np
import sys
sys.path.append("..")
from data_processing.train_test_split import train_test_split
from models.knn import KNearestNeighbours 
from data_processing.preprocessing import MinMaxScaler

# Load Ionosphere dataset
iono = np.genfromtxt("../datasets/ionosphere.txt", delimiter=',')
iono_X = iono[:, :-1]  # Features
iono_y = iono[:, -1]   # Labels

# Split the dataset
X_train, X_test, y_train, y_test = train_test_split(iono_X, iono_y, test_size=0.2, seed=42)

# Initialize KNN classifier
knn = KNearestNeighbours(3)  

# Training and testing without scaling
knn.fit(X_train, y_train)
predictions = knn.predict(X_test)
accuracy_without_scaling = np.mean(predictions == y_test)

# Apply MinMaxScaler
scaler = MinMaxScaler()
scaled_X_train = scaler.fit_transform(X_train)
scaled_X_test = scaler.transform(X_test)

# Training and testing with scaling
knn.fit(scaled_X_train, y_train)
scaled_predictions = knn.predict(scaled_X_test)
accuracy_with_scaling = np.mean(scaled_predictions == y_test)

# Print out the results
print(f"Accuracy without scaling: {accuracy_without_scaling * 100:.2f}%")
print(f"Accuracy with scaling: {accuracy_with_scaling * 100:.2f}%")


Accuracy without scaling: 87.14%
Accuracy with scaling: 87.14%


In [2]:
print(X_train)

[[ 1.       0.       0.4709  ... -0.24339  0.2672   0.04233]
 [ 1.       0.       0.82254 ... -0.0422   0.78439  0.01214]
 [ 1.       0.       0.89589 ... -0.07029  0.76862  0.27926]
 ...
 [ 0.       0.       1.      ...  1.       1.      -1.     ]
 [ 1.       0.       0.97905 ...  0.18345 -0.64134  0.02968]
 [ 1.       0.       0.97032 ... -0.08676  0.96575 -0.21918]]


In [3]:
print(scaled_X_train)

[[1.       0.       0.73545  ... 0.378305 0.6336   0.521165]
 [1.       0.       0.91127  ... 0.4789   0.892195 0.50607 ]
 [1.       0.       0.947945 ... 0.464855 0.88431  0.63963 ]
 ...
 [0.       0.       1.       ... 1.       1.       0.      ]
 [1.       0.       0.989525 ... 0.591725 0.17933  0.51484 ]
 [1.       0.       0.98516  ... 0.45662  0.982875 0.39041 ]]


In [4]:
from sklearn.datasets import load_iris

# Load Iris dataset
iris = load_iris()
iris_X = iris.data
iris_y = iris.target

# Split the dataset
X_train, X_test, y_train, y_test = train_test_split(iris_X, iris_y, test_size=0.2, seed=42)

# Initialize KNN classifier
knn = KNearestNeighbours(10)  

# Training and testing without scaling
knn.fit(X_train, y_train)
predictions = knn.predict(X_test)
accuracy_without_scaling = np.mean(predictions == y_test)

# Apply MinMaxScaler
scaler = MinMaxScaler()
scaled_X_train = scaler.fit_transform(X_train)
scaled_X_test = scaler.transform(X_test)

# Training and testing with scaling
knn.fit(scaled_X_train, y_train)
scaled_predictions = knn.predict(scaled_X_test)
accuracy_with_scaling = np.mean(scaled_predictions == y_test)

# Print out the results
print(f"Accuracy without scaling: {accuracy_without_scaling * 100:.2f}%")
print(f"Accuracy with scaling: {accuracy_with_scaling * 100:.2f}%")

Accuracy without scaling: 96.67%
Accuracy with scaling: 96.67%


In [5]:
print(X_train)

[[6.1 2.8 4.7 1.2]
 [5.7 3.8 1.7 0.3]
 [7.7 2.6 6.9 2.3]
 [6.  2.9 4.5 1.5]
 [6.8 2.8 4.8 1.4]
 [5.4 3.4 1.5 0.4]
 [5.6 2.9 3.6 1.3]
 [6.9 3.1 5.1 2.3]
 [6.2 2.2 4.5 1.5]
 [5.8 2.7 3.9 1.2]
 [6.5 3.2 5.1 2. ]
 [4.8 3.  1.4 0.1]
 [5.5 3.5 1.3 0.2]
 [4.9 3.1 1.5 0.1]
 [5.1 3.8 1.5 0.3]
 [6.3 3.3 4.7 1.6]
 [6.5 3.  5.8 2.2]
 [5.6 2.5 3.9 1.1]
 [5.7 2.8 4.5 1.3]
 [6.4 2.8 5.6 2.2]
 [4.7 3.2 1.6 0.2]
 [6.1 3.  4.9 1.8]
 [5.  3.4 1.6 0.4]
 [6.4 2.8 5.6 2.1]
 [7.9 3.8 6.4 2. ]
 [6.7 3.  5.2 2.3]
 [6.7 2.5 5.8 1.8]
 [6.8 3.2 5.9 2.3]
 [4.8 3.  1.4 0.3]
 [4.8 3.1 1.6 0.2]
 [4.6 3.6 1.  0.2]
 [5.7 4.4 1.5 0.4]
 [6.7 3.1 4.4 1.4]
 [4.8 3.4 1.6 0.2]
 [4.4 3.2 1.3 0.2]
 [6.3 2.5 5.  1.9]
 [6.4 3.2 4.5 1.5]
 [5.2 3.5 1.5 0.2]
 [5.  3.6 1.4 0.2]
 [5.2 4.1 1.5 0.1]
 [5.8 2.7 5.1 1.9]
 [6.  3.4 4.5 1.6]
 [6.7 3.1 4.7 1.5]
 [5.4 3.9 1.3 0.4]
 [5.4 3.7 1.5 0.2]
 [5.5 2.4 3.7 1. ]
 [6.3 2.8 5.1 1.5]
 [6.4 3.1 5.5 1.8]
 [6.6 3.  4.4 1.4]
 [7.2 3.6 6.1 2.5]
 [5.7 2.9 4.2 1.3]
 [7.6 3.  6.6 2.1]
 [5.6 3.  4.

In [6]:
print(scaled_X_train)

[[0.5        0.33333333 0.62711864 0.45833333]
 [0.38888889 0.75       0.11864407 0.08333333]
 [0.94444444 0.25       1.         0.91666667]
 [0.47222222 0.375      0.59322034 0.58333333]
 [0.69444444 0.33333333 0.6440678  0.54166667]
 [0.30555556 0.58333333 0.08474576 0.125     ]
 [0.36111111 0.375      0.44067797 0.5       ]
 [0.72222222 0.45833333 0.69491525 0.91666667]
 [0.52777778 0.08333333 0.59322034 0.58333333]
 [0.41666667 0.29166667 0.49152542 0.45833333]
 [0.61111111 0.5        0.69491525 0.79166667]
 [0.13888889 0.41666667 0.06779661 0.        ]
 [0.33333333 0.625      0.05084746 0.04166667]
 [0.16666667 0.45833333 0.08474576 0.        ]
 [0.22222222 0.75       0.08474576 0.08333333]
 [0.55555556 0.54166667 0.62711864 0.625     ]
 [0.61111111 0.41666667 0.81355932 0.875     ]
 [0.36111111 0.20833333 0.49152542 0.41666667]
 [0.38888889 0.33333333 0.59322034 0.5       ]
 [0.58333333 0.33333333 0.77966102 0.875     ]
 [0.11111111 0.5        0.10169492 0.04166667]
 [0.5        