In [4]:
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import LabelEncoder
from sklearn.impute import SimpleImputer
from sklearn.ensemble import RandomForestClassifier
from sklearn.metrics import accuracy_score, confusion_matrix, classification_report

# Load the dataset
titanic_data_path = 'D:\Afame\Titanic-Dataset.csv'
titanic_df = pd.read_csv(titanic_data_path)

# Display the first few rows of the dataset
print("Original Data:")
print(titanic_df.head())

# Basic information about the dataset
print(titanic_df.info())

# Summary statistics
print(titanic_df.describe())

# Check for missing values
print(titanic_df.isnull().sum())

# Handle missing values for 'Age' with SimpleImputer
imputer_age = SimpleImputer(strategy='mean')
titanic_df['Age'] = imputer_age.fit_transform(titanic_df[['Age']])

# Encode categorical variable 'Embarked' before imputing missing values
titanic_df['Embarked'] = titanic_df['Embarked'].fillna('S')
label_encoder = LabelEncoder()
titanic_df['Embarked'] = label_encoder.fit_transform(titanic_df['Embarked'])

# Drop unnecessary columns
titanic_df = titanic_df.drop(columns=['Cabin', 'Ticket', 'Name', 'PassengerId'])

# Encode other categorical variables
titanic_df['Sex'] = label_encoder.fit_transform(titanic_df['Sex'])

# Display the cleaned dataset
print("Cleaned Data:")
print(titanic_df.head())

# Define features and target variable
X = titanic_df.drop(columns=['Survived'])
y = titanic_df['Survived']

# Split the data into training and testing sets
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Initialize the model
model = RandomForestClassifier(n_estimators=100, random_state=42)

# Train the model
model.fit(X_train, y_train)

# Make predictions
y_pred = model.predict(X_test)

# Evaluate the model
print("Accuracy:", accuracy_score(y_test, y_pred))
print("Confusion Matrix:")
print(confusion_matrix(y_test, y_pred))
print("Classification Report:")
print(classification_report(y_test, y_pred))

# Make predictions on new data
# Example: [Pclass, Sex, Age, SibSp, Parch, Fare, Embarked]
new_data = [[3, 1, 22.0, 1, 0, 7.25, 2]]  
new_data_df = pd.DataFrame(new_data, columns=X.columns)
prediction = model.predict(new_data_df)
print("Survived" if prediction[0] == 1 else "Not Survived")


Original Data:
   PassengerId  Survived  Pclass  \
0            1         0       3   
1            2         1       1   
2            3         1       3   
3            4         1       1   
4            5         0       3   

                                                Name     Sex   Age  SibSp  \
0                            Braund, Mr. Owen Harris    male  22.0      1   
1  Cumings, Mrs. John Bradley (Florence Briggs Th...  female  38.0      1   
2                             Heikkinen, Miss. Laina  female  26.0      0   
3       Futrelle, Mrs. Jacques Heath (Lily May Peel)  female  35.0      1   
4                           Allen, Mr. William Henry    male  35.0      0   

   Parch            Ticket     Fare Cabin Embarked  
0      0         A/5 21171   7.2500   NaN        S  
1      0          PC 17599  71.2833   C85        C  
2      0  STON/O2. 3101282   7.9250   NaN        S  
3      0            113803  53.1000  C123        S  
4      0            373450   8.0500   Na