In [None]:

# 1. Import Required Libraries
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns

from sklearn.preprocessing import LabelEncoder, MinMaxScaler, StandardScaler
from sklearn.model_selection import train_test_split
from sklearn.ensemble import RandomForestClassifier
from sklearn.svm import SVC
from sklearn.metrics import classification_report, confusion_matrix, accuracy_score


In [None]:

# 2. Load the Dataset
df = pd.read_csv('/content/heart.csv')  # Change path if needed
print(df.head())
print(df.info())


In [None]:

# 3. Handle Missing Values
print("\nMissing Values Before:")
print(df.isnull().sum())

df = df.dropna()

print("\nMissing Values After:")
print(df.isnull().sum())


In [None]:

# 4. Remove Duplicate Rows
print("\nDuplicate Rows Before:", df.duplicated().sum())
df = df.drop_duplicates()
print("Duplicate Rows After:", df.duplicated().sum())


In [None]:

# 5. Exploratory Data Analysis (EDA)

# 5.1 Distribution of each feature
for column in df.columns:
    plt.figure(figsize=(6, 4))
    sns.distplot(df[column], kde=True)
    plt.title(f'Distribution of {column}')
    plt.show()


In [None]:

# 5.2 Boxplots to detect Outliers
for column in df.select_dtypes(include=np.number).columns:
    plt.figure(figsize=(6, 4))
    sns.boxplot(x=df[column])
    plt.title(f'Boxplot of {column}')
    plt.show()


In [None]:

# 5.3 Pairplot
sns.pairplot(df, hue='target')
plt.show()


In [None]:

# 5.4 Correlation Matrix Heatmap
plt.figure(figsize=(12,8))
sns.heatmap(df.corr(), annot=True, cmap='coolwarm')
plt.title('Correlation Matrix')
plt.show()


In [None]:

# 6. Encoding Categorical Variables if needed
print(df.dtypes)

label_encoders = {}
for column in df.select_dtypes(include='object').columns:
    le = LabelEncoder()
    df[column] = le.fit_transform(df[column])
    label_encoders[column] = le


In [None]:

# 7. Normalization
scaler = StandardScaler()
scaled_features = scaler.fit_transform(df.drop('target', axis=1))
df_scaled = pd.DataFrame(scaled_features, columns=df.drop('target', axis=1).columns)


In [None]:

# 8. Define Feature Variables and Target Variable
X = df_scaled
y = df['target']

print("\nFeature Variables (X):")
print(X.head())
print("\nTarget Variable (y):")
print(y.head())


In [None]:

# 9. Train-Test Split
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)


In [None]:

# 10. Apply Two Different Models

# Model 1: Random Forest Classifier
rf_model = RandomForestClassifier(random_state=42)
rf_model.fit(X_train, y_train)
rf_pred = rf_model.predict(X_test)

# Model 2: Support Vector Machine Classifier
svm_model = SVC(random_state=42)
svm_model.fit(X_train, y_train)
svm_pred = svm_model.predict(X_test)


In [None]:

# 11. Evaluate Both Models
print("\nRandom Forest Classifier Results:")
print(confusion_matrix(y_test, rf_pred))
print(classification_report(y_test, rf_pred))
print("Accuracy:", accuracy_score(y_test, rf_pred))

print("\nSupport Vector Machine Results:")
print(confusion_matrix(y_test, svm_pred))
print(classification_report(y_test, svm_pred))
print("Accuracy:", accuracy_score(y_test, svm_pred))
