In [2]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.ensemble import RandomForestClassifier
from sklearn.metrics import accuracy_score, classification_report, confusion_matrix

# Load Titanic dataset from CSV file
# Replace 'path_to_your_file/titanic.csv' with the actual path of your file
df = pd.read_csv('/home/rgukt/Downloads/titanic.csv')

# Display the first few rows of the dataset
print("First 5 rows of the Titanic dataset:")
print(df.head())

# Handle missing values
df['Age'] = df['Age'].fillna(df['Age'].median())  # Avoid inplace=True by reassigning to the column
df['Embarked'] = df['Embarked'].fillna(df['Embarked'].mode()[0])  # Reassign the filled values

# Drop columns that won't be used or have too many missing values
df.drop(columns=['Cabin', 'Ticket', 'Name'], inplace=True)  # We can still use inplace here without warnings

# Convert categorical variables (e.g., 'Sex' and 'Embarked') to numeric using one-hot encoding
df = pd.get_dummies(df, drop_first=True)

# Display the updated dataset after preprocessing
print("\nDataset after preprocessing:")
print(df.head())


# Features (X) and target (y)
X = df.drop(columns='Survived')
y = df['Survived']

# Split the dataset into training and testing sets (80% training, 20% testing)
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Train a Random Forest Classifier
model = RandomForestClassifier(n_estimators=100, random_state=42)
model.fit(X_train, y_train)

# Make predictions on the test set
y_pred = model.predict(X_test)

# Evaluate the model
print(f'\nAccuracy: {accuracy_score(y_test, y_pred):.2f}')
print("\nClassification Report:")
print(classification_report(y_test, y_pred))

# Confusion Matrix
print("\nConfusion Matrix:")
print(confusion_matrix(y_test, y_pred))


First 5 rows of the Titanic dataset:
   PassengerId  Survived  Pclass  \
0            1         0       3   
1            2         1       1   
2            3         1       3   
3            4         1       1   
4            5         0       3   

                                                Name     Sex   Age  SibSp  \
0                            Braund, Mr. Owen Harris    male  22.0      1   
1  Cumings, Mrs. John Bradley (Florence Briggs Th...  female  38.0      1   
2                             Heikkinen, Miss. Laina  female  26.0      0   
3       Futrelle, Mrs. Jacques Heath (Lily May Peel)  female  35.0      1   
4                           Allen, Mr. William Henry    male  35.0      0   

   Parch            Ticket     Fare Cabin Embarked  
0      0         A/5 21171   7.2500   NaN        S  
1      0          PC 17599  71.2833   C85        C  
2      0  STON/O2. 3101282   7.9250   NaN        S  
3      0            113803  53.1000  C123        S  
4      0          