In [1]:
import numpy as np
import pandas as pd

from sklearn.model_selection import train_test_split 
from sklearn.impute import SimpleImputer 
from sklearn.preprocessing import OneHotEncoder
from sklearn.preprocessing import MinMaxScaler 
from sklearn.tree import DecisionTreeClassifier 

In [2]:
titanic_dataset = pd.read_csv('train.csv')

In [3]:
titanic_dataset.sample(5)

Unnamed: 0,PassengerId,Survived,Pclass,Name,Sex,Age,SibSp,Parch,Ticket,Fare,Cabin,Embarked
394,395,1,3,"Sandstrom, Mrs. Hjalmar (Agnes Charlotta Bengt...",female,24.0,0,2,PP 9549,16.7,G6,S
739,740,0,3,"Nankoff, Mr. Minko",male,,0,0,349218,7.8958,,S
158,159,0,3,"Smiljanic, Mr. Mile",male,,0,0,315037,8.6625,,S
742,743,1,1,"Ryerson, Miss. Susan Parker ""Suzette""",female,21.0,2,2,PC 17608,262.375,B57 B59 B63 B66,C
434,435,0,1,"Silvey, Mr. William Baird",male,50.0,1,0,13507,55.9,E44,S


In [4]:
# here we working for the pipeline in the dataframe not working for making the best model with dataframe 
titanic_dataset =titanic_dataset.drop(columns = ['PassengerId','Name','Ticket','Cabin'])

In [5]:
titanic_dataset 

Unnamed: 0,Survived,Pclass,Sex,Age,SibSp,Parch,Fare,Embarked
0,0,3,male,22.0,1,0,7.2500,S
1,1,1,female,38.0,1,0,71.2833,C
2,1,3,female,26.0,0,0,7.9250,S
3,1,1,female,35.0,1,0,53.1000,S
4,0,3,male,35.0,0,0,8.0500,S
...,...,...,...,...,...,...,...,...
886,0,2,male,27.0,0,0,13.0000,S
887,1,1,female,19.0,0,0,30.0000,S
888,0,3,female,,1,2,23.4500,S
889,1,1,male,26.0,0,0,30.0000,C


In [6]:
# steps 1 --> train/test/split
x_train,x_test,y_train,y_test = train_test_split(titanic_dataset.iloc[:,1:],titanic_dataset.iloc[:,-8],test_size = 0.2,random_state = 42)

In [7]:
x_train.head(2)

Unnamed: 0,Pclass,Sex,Age,SibSp,Parch,Fare,Embarked
331,1,male,45.5,0,0,28.5,S
733,2,male,23.0,0,0,13.0,S


In [8]:
y_train.head()

331    0
733    0
382    0
704    0
813    0
Name: Survived, dtype: int64

In [9]:
titanic_dataset.isnull().sum()

Survived      0
Pclass        0
Sex           0
Age         177
SibSp         0
Parch         0
Fare          0
Embarked      2
dtype: int64

In [10]:
# Applying imputation 

si_age = SimpleImputer()
si_embarked = SimpleImputer(strategy = 'most_frequent')
# staratgy = 'most_frequent'  ------> it mean that which is mostly replating in the dataframe(will be replace by the nan values )
x_train_age = si_age.fit_transform(x_train[['Age']])
x_train_embarked = si_embarked.fit_transform(x_train[['Embarked']])

x_test_age = si_age = si_age.transform(x_test[['Age']])
x_test_embarked = si_embarked.transform(x_test[['Embarked']])

In [11]:
# one hot encoding sex and Embarked 

ohe_sex = OneHotEncoder(sparse =False,handle_unknown = 'ignore') # here sparse = false -----> it for the givneing the array 
ohe_embarked = OneHotEncoder(sparse = False,handle_unknown = 'ignore')

x_train_sex =ohe_sex.fit_transform(x_train[['Sex']])
x_train_embarked = ohe_embarked.fit_transform(x_train[['Embarked']])

x_test_sex = ohe_sex.transform(x_test[['Sex']])
x_test_embarked = ohe_embarked.transform(x_test[['Embarked']])



In [12]:
x_train_embarked
x_train_embarked.shape

(712, 4)

In [13]:
x_train_age.shape

(712, 1)

In [14]:

x_train_sex
x_train_sex.shape

(712, 2)

In [15]:
x_test_embarked
x_test_embarked.shape

(179, 4)

In [16]:
# those how appled the imputetion and ecoding we drop it 
x_train.head(2)

Unnamed: 0,Pclass,Sex,Age,SibSp,Parch,Fare,Embarked
331,1,male,45.5,0,0,28.5,S
733,2,male,23.0,0,0,13.0,S


In [17]:
x_train_rem = x_train.drop(columns =['Sex','Age','Embarked'],axis =1)

In [18]:
# x_train_rem = x_train_rem.to_numpy()

In [19]:
x_train_rem.shape

(712, 4)

In [20]:
x_test_rem = x_test.drop(columns=['Sex','Age','Embarked'],axis = 1)

In [21]:
# x_test_rem= x_test_rem.to_numpy()

In [22]:
x_test_rem.shape

(179, 4)

In [23]:
x_train_transformed = np.concatenate((x_train_rem,x_train_age,x_train_sex,x_train_embarked),axis = 1)
x_test_transformed = np.concatenate((x_test_rem,x_test_age,x_test_sex,x_test_embarked),axis =1)

In [24]:
x_train_transformed.shape

(712, 11)

In [25]:
x_train_transformed

array([[1., 0., 0., ..., 0., 1., 0.],
       [2., 0., 0., ..., 0., 1., 0.],
       [3., 0., 0., ..., 0., 1., 0.],
       ...,
       [3., 2., 0., ..., 0., 1., 0.],
       [1., 1., 2., ..., 0., 1., 0.],
       [1., 0., 1., ..., 0., 1., 0.]])

In [26]:
# here we creating the classfier 
clf = DecisionTreeClassifier()
clf.fit(x_train_transformed,y_train)

In [27]:
y_pred = clf.predict(x_test_transformed)
y_pred

array([0, 1, 0, 1, 1, 0, 1, 0, 1, 1, 0, 0, 0, 0, 0, 1, 0, 1, 0, 0, 0, 0,
       0, 0, 0, 0, 0, 1, 0, 0, 0, 1, 1, 1, 0, 0, 1, 1, 1, 0, 0, 1, 0, 0,
       0, 0, 0, 0, 1, 0, 1, 1, 0, 1, 1, 1, 1, 0, 1, 0, 1, 1, 1, 0, 0, 1,
       0, 0, 0, 1, 1, 1, 1, 1, 0, 0, 1, 1, 1, 1, 0, 1, 1, 0, 0, 0, 1, 1,
       0, 0, 1, 0, 0, 0, 0, 0, 1, 0, 1, 0, 0, 0, 1, 0, 0, 1, 1, 0, 0, 0,
       1, 0, 1, 1, 0, 0, 0, 0, 1, 0, 0, 1, 1, 1, 0, 0, 1, 1, 0, 0, 1, 0,
       0, 1, 1, 0, 1, 0, 0, 0, 0, 1, 0, 0, 0, 1, 0, 1, 1, 0, 0, 1, 0, 0,
       0, 0, 0, 1, 1, 1, 0, 0, 0, 1, 0, 0, 0, 1, 0, 0, 0, 0, 1, 1, 0, 0,
       0, 1, 1], dtype=int64)

In [28]:
from sklearn.metrics import accuracy_score 
accuracy_score(y_test,y_pred)

0.776536312849162

In [29]:
import pickle 

In [32]:
import os
import pickle

# Create the 'models' directory if it doesn't exist
if not os.path.exists('models'):
    os.makedirs('models')

# Assuming ohe_sex, ohe_embarked, and clf are already defined
pickle.dump(ohe_sex, open('models/ohe_sex.pkl', 'wb'))   # here we dowing the ecoding the model in the dataframe
pickle.dump(ohe_embarked, open('models/ohe_embarked.pkl', 'wb'))
pickle.dump(clf, open('models/clf.pkl', 'wb'))
