In [1]:
import pandas as pd
import numpy as np
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LogisticRegression
from sklearn.ensemble import RandomForestClassifier
from sklearn.svm import SVC
from sklearn.preprocessing import StandardScaler, LabelEncoder
from sklearn.metrics import accuracy_score, confusion_matrix, classification_report

# Load the dataset
url = "https://raw.githubusercontent.com/FlipRoboTechnologies/ML-Datasets/main/Titanic/titanic_train.csv"
data = pd.read_csv(url)

# Display the first few rows of the dataset
print(data.head())

# Display basic information about the dataset
print(data.info())

# Check for missing values
print(data.isnull().sum())

# Summary statistics
print(data.describe())

# Handle missing values
data['Age'].fillna(data['Age'].median(), inplace=True)
data['Embarked'].fillna(data['Embarked'].mode()[0], inplace=True)
data['Fare'].fillna(data['Fare'].median(), inplace=True)

# Drop 'Cabin' because it has too many missing values
data.drop('Cabin', axis=1, inplace=True)

# Drop 'Name' and 'Ticket' as they are not useful for prediction
data.drop(['Name', 'Ticket'], axis=1, inplace=True)

# Encode categorical variables
labelencoder = LabelEncoder()
data['Sex'] = labelencoder.fit_transform(data['Sex'])
data['Embarked'] = labelencoder.fit_transform(data['Embarked'])

# Extract features and target variable
X = data.drop('Survived', axis=1)
y = data['Survived']

# Scale the features
scaler = StandardScaler()
X_scaled = scaler.fit_transform(X)

# Split the data into training and testing sets
X_train, X_test, y_train, y_test = train_test_split(X_scaled, y, test_size=0.2, random_state=42)

# Initialize the models
log_reg = LogisticRegression()
rf_clf = RandomForestClassifier(random_state=42)
svc = SVC()

# Train and evaluate Logistic Regression
log_reg.fit(X_train, y_train)
log_reg_pred = log_reg.predict(X_test)
log_reg_acc = accuracy_score(y_test, log_reg_pred)
print(f'Logistic Regression Accuracy: {log_reg_acc}')
print(confusion_matrix(y_test, log_reg_pred))
print(classification_report(y_test, log_reg_pred))

# Train and evaluate Random Forest Classifier
rf_clf.fit(X_train, y_train)
rf_clf_pred = rf_clf.predict(X_test)
rf_clf_acc = accuracy_score(y_test, rf_clf_pred)
print(f'Random Forest Accuracy: {rf_clf_acc}')
print(confusion_matrix(y_test, rf_clf_pred))
print(classification_report(y_test, rf_clf_pred))

# Train and evaluate Support Vector Classifier
svc.fit(X_train, y_train)
svc_pred = svc.predict(X_test)
svc_acc = accuracy_score(y_test, svc_pred)
print(f'Support Vector Classifier Accuracy: {svc_acc}')
print(confusion_matrix(y_test, svc_pred))
print(classification_report(y_test, svc_pred))


   PassengerId  Survived  Pclass  \
0            1         0       3   
1            2         1       1   
2            3         1       3   
3            4         1       1   
4            5         0       3   

                                                Name     Sex   Age  SibSp  \
0                            Braund, Mr. Owen Harris    male  22.0      1   
1  Cumings, Mrs. John Bradley (Florence Briggs Th...  female  38.0      1   
2                             Heikkinen, Miss. Laina  female  26.0      0   
3       Futrelle, Mrs. Jacques Heath (Lily May Peel)  female  35.0      1   
4                           Allen, Mr. William Henry    male  35.0      0   

   Parch            Ticket     Fare Cabin Embarked  
0      0         A/5 21171   7.2500   NaN        S  
1      0          PC 17599  71.2833   C85        C  
2      0  STON/O2. 3101282   7.9250   NaN        S  
3      0            113803  53.1000  C123        S  
4      0            373450   8.0500   NaN        S  
<c