# Handling Categorical Data | One Hot Encoding

In [1]:
import numpy as np
import pandas as pd

In [2]:
df = pd.read_csv('cars.csv')

In [3]:
df.head()

Unnamed: 0,brand,km_driven,fuel,owner,selling_price
0,Maruti,145500,Diesel,First Owner,450000
1,Skoda,120000,Diesel,Second Owner,370000
2,Honda,140000,Petrol,Third Owner,158000
3,Hyundai,127000,Diesel,First Owner,225000
4,Maruti,120000,Petrol,First Owner,130000


In [4]:
df['owner'].value_counts()

owner
First Owner             5289
Second Owner            2105
Third Owner              555
Fourth & Above Owner     174
Test Drive Car             5
Name: count, dtype: int64

### 1. One hot encoding using pandas 

In [5]:
pd.get_dummies(df,columns=['fuel','owner'])

Unnamed: 0,brand,km_driven,selling_price,fuel_CNG,fuel_Diesel,fuel_LPG,fuel_Petrol,owner_First Owner,owner_Fourth & Above Owner,owner_Second Owner,owner_Test Drive Car,owner_Third Owner
0,Maruti,145500,450000,False,True,False,False,True,False,False,False,False
1,Skoda,120000,370000,False,True,False,False,False,False,True,False,False
2,Honda,140000,158000,False,False,False,True,False,False,False,False,True
3,Hyundai,127000,225000,False,True,False,False,True,False,False,False,False
4,Maruti,120000,130000,False,False,False,True,True,False,False,False,False
...,...,...,...,...,...,...,...,...,...,...,...,...
8123,Hyundai,110000,320000,False,False,False,True,True,False,False,False,False
8124,Hyundai,119000,135000,False,True,False,False,False,True,False,False,False
8125,Maruti,120000,382000,False,True,False,False,True,False,False,False,False
8126,Tata,25000,290000,False,True,False,False,True,False,False,False,False


### 2. K-1 one hot encoding

In [6]:
pd.get_dummies(df,columns=['fuel','owner'],drop_first=True)

Unnamed: 0,brand,km_driven,selling_price,fuel_Diesel,fuel_LPG,fuel_Petrol,owner_Fourth & Above Owner,owner_Second Owner,owner_Test Drive Car,owner_Third Owner
0,Maruti,145500,450000,True,False,False,False,False,False,False
1,Skoda,120000,370000,True,False,False,False,True,False,False
2,Honda,140000,158000,False,False,True,False,False,False,True
3,Hyundai,127000,225000,True,False,False,False,False,False,False
4,Maruti,120000,130000,False,False,True,False,False,False,False
...,...,...,...,...,...,...,...,...,...,...
8123,Hyundai,110000,320000,False,False,True,False,False,False,False
8124,Hyundai,119000,135000,True,False,False,True,False,False,False
8125,Maruti,120000,382000,True,False,False,False,False,False,False
8126,Tata,25000,290000,True,False,False,False,False,False,False


### 3. One hot encoding using sklearn

In [7]:
X = df.drop(['selling_price'], axis=1)
y = df['selling_price']

In [8]:
from sklearn.model_selection import train_test_split
X_train,X_test,y_train,y_test = train_test_split(X,y, test_size= 0.2)

In [9]:
X_train.head()

Unnamed: 0,brand,km_driven,fuel,owner
278,Tata,120000,Diesel,First Owner
5335,Hyundai,74000,Diesel,Second Owner
8077,Toyota,250000,Diesel,First Owner
5092,Toyota,79328,Diesel,Second Owner
5482,Mahindra,80000,Diesel,Second Owner


In [10]:
from sklearn.preprocessing import OneHotEncoder

In [12]:
ohe = OneHotEncoder(drop='first',sparse_output=False,dtype=np.int32)

In [13]:
X_train_new = ohe.fit_transform(X_train[['fuel','owner']])

In [14]:
X_test_new = ohe.transform(X_test[['fuel','owner']])

In [15]:
X_train_new.shape

(6502, 7)

In [16]:
np.hstack((X_train[['brand','km_driven']].values,X_train_new))

array([['Tata', 120000, 1, ..., 0, 0, 0],
       ['Hyundai', 74000, 1, ..., 1, 0, 0],
       ['Toyota', 250000, 1, ..., 0, 0, 0],
       ...,
       ['Hyundai', 60000, 0, ..., 1, 0, 0],
       ['Honda', 110000, 0, ..., 1, 0, 0],
       ['Jaguar', 45000, 1, ..., 0, 0, 0]], dtype=object)

### 4. One hot encoding with top categories

In [17]:
counts = df['brand'].value_counts()

In [18]:
df['brand'].nunique()
threshold = 100

In [19]:
repl = counts[counts <= threshold].index

In [20]:
pd.get_dummies(df['brand'].replace(repl, 'uncommon')).sample(5)

Unnamed: 0,BMW,Chevrolet,Ford,Honda,Hyundai,Mahindra,Maruti,Renault,Skoda,Tata,Toyota,Volkswagen,uncommon
3300,False,False,False,False,False,False,True,False,False,False,False,False,False
5990,False,False,False,False,False,False,False,False,False,False,False,False,True
945,False,False,False,False,False,False,True,False,False,False,False,False,False
3075,False,False,False,False,True,False,False,False,False,False,False,False,False
6772,False,False,True,False,False,False,False,False,False,False,False,False,False
