# Titanic Survival Prediction
## 1. Import Libraries
Importing necessary libraries for data handling and machine learning.

In [4]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.tree import DecisionTreeClassifier
from sklearn.metrics import accuracy_score

## 2. Load Dataset
Loading the Titanic dataset.

In [5]:
# Load the dataset using the file path
# Note: Ensure that the path is correct based on your local computer
address = r"C:\Users\ND.COM\Desktop\Titanic\train.csv"
df = pd.read_csv(address)

# Display the first few rows
print(df.head())

   PassengerId  Survived  Pclass  \
0            1         0       3   
1            2         1       1   
2            3         1       3   
3            4         1       1   
4            5         0       3   

                                                Name     Sex   Age  SibSp  \
0                            Braund, Mr. Owen Harris    male  22.0      1   
1  Cumings, Mrs. John Bradley (Florence Briggs Th...  female  38.0      1   
2                             Heikkinen, Miss. Laina  female  26.0      0   
3       Futrelle, Mrs. Jacques Heath (Lily May Peel)  female  35.0      1   
4                           Allen, Mr. William Henry    male  35.0      0   

   Parch            Ticket     Fare Cabin Embarked  
0      0         A/5 21171   7.2500   NaN        S  
1      0          PC 17599  71.2833   C85        C  
2      0  STON/O2. 3101282   7.9250   NaN        S  
3      0            113803  53.1000  C123        S  
4      0            373450   8.0500   NaN        S  


## 3. Data Cleaning and Preprocessing
Here, I perform the following steps:
1. Drop the irrelevant columns (PassengerId, Name, Cabin, Ticket, Embarked).
2. Handle the missing values in the 'Age' column.
3. Convert the categorical 'Sex' column to numerical values.

In [6]:
# Drop the columns that are not useful for prediction
df = df.drop(['PassengerId', 'Name', 'Cabin', 'Ticket', 'Embarked'], axis=1)

# Filling missing values in 'Age' with the mean (average) age
avg_age = df['Age'].mean()
df['Age'] = df['Age'].fillna(avg_age)

# Converting 'Sex' column to numerical: for Male = 0, for Female = 1
df['Sex'] = df['Sex'].map({'male': 0, 'female': 1})

# Checking the cleaned data
print(df.info())

<class 'pandas.core.frame.DataFrame'>
RangeIndex: 891 entries, 0 to 890
Data columns (total 7 columns):
 #   Column    Non-Null Count  Dtype  
---  ------    --------------  -----  
 0   Survived  891 non-null    int64  
 1   Pclass    891 non-null    int64  
 2   Sex       891 non-null    int64  
 3   Age       891 non-null    float64
 4   SibSp     891 non-null    int64  
 5   Parch     891 non-null    int64  
 6   Fare      891 non-null    float64
dtypes: float64(2), int64(5)
memory usage: 48.9 KB
None


## 4. Feature Selection
Separating the data into Features (X) and Target (y).

In [7]:
# X contains the features (input)
X = df.drop('Survived', axis=1)

# y contains the target variable (output)
y = df['Survived']

## 5. Model Training
Splitting the data into training and testing sets, then training the Decision Tree model.

In [8]:
# Splitting data: 80% for training, 20% for testing
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Initializing the Decision Tree Classifier
model = DecisionTreeClassifier()

# Training model
model.fit(X_train, y_train)

## 6. Evaluation
Testing the model accuracy on unseen data.

In [9]:
# Making predictions on the test set
predictions = model.predict(X_test)

# Calculating accuracy
accuracy = accuracy_score(y_test, predictions)

# Printing the final accuracy of model
print(f"Model Accuracy: {accuracy * 100:.2f}%")

Model Accuracy: 74.30%
