# EDA

In [27]:
import pandas as pd

# 데이터 로드
url = 'https://raw.githubusercontent.com/datasciencedojo/datasets/master/titanic.csv'
df = pd.read_csv(url)

# 기본 정보 확인
print(df.head())
print(df.info())
print(df.describe())

# 결측치 확인
print("결측치 수:\n", df.isnull().sum())

# Survived 클래스별 분포
print("생존자 수:\n", df['Survived'].value_counts())


   PassengerId  Survived  Pclass  \
0            1         0       3   
1            2         1       1   
2            3         1       3   
3            4         1       1   
4            5         0       3   

                                                Name     Sex   Age  SibSp  \
0                            Braund, Mr. Owen Harris    male  22.0      1   
1  Cumings, Mrs. John Bradley (Florence Briggs Th...  female  38.0      1   
2                             Heikkinen, Miss. Laina  female  26.0      0   
3       Futrelle, Mrs. Jacques Heath (Lily May Peel)  female  35.0      1   
4                           Allen, Mr. William Henry    male  35.0      0   

   Parch            Ticket     Fare Cabin Embarked  
0      0         A/5 21171   7.2500   NaN        S  
1      0          PC 17599  71.2833   C85        C  
2      0  STON/O2. 3101282   7.9250   NaN        S  
3      0            113803  53.1000  C123        S  
4      0            373450   8.0500   NaN        S  
<c

# Preprocessing

In [32]:
# Age 결측치는 중앙값으로 채우기
df['Age'] = df['Age'].fillna(df['Age'].median())

# Embarked 결측치는 최빈값으로 채우기
df['Embarked'] = df['Embarked'].fillna(df['Embarked'].mode()[0])

# Sex, Embarked → 숫자 인코딩
df['Sex'] = df['Sex'].map({'male': 0, 'female': 1})
df['Embarked'] = df['Embarked'].map({'S': 0, 'C': 1, 'Q': 2})

# 필요 없는 컬럼 제거
df.drop(columns=['Name', 'Ticket', 'Cabin'], inplace=True)

KeyError: "['Name', 'Ticket', 'Cabin'] not found in axis"

# 피처와 타겟 설정

In [29]:
from sklearn.model_selection import train_test_split

# 피처와 타겟 설정
X = df[['Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare', 'Embarked']]
y = df['Survived']

# 학습/테스트 데이터 분할
X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42, stratify=y
)

# 모델 훈련 및 평가

In [30]:
from sklearn.linear_model import LogisticRegression
from sklearn.naive_bayes import GaussianNB
from sklearn.metrics import accuracy_score, confusion_matrix

# 로지스틱 회귀
logreg = LogisticRegression(solver='liblinear')
logreg.fit(X_train, y_train)
logreg_preds = logreg.predict(X_test)
print("Logistic Regression 정확도:", round(accuracy_score(y_test, logreg_preds), 3))
print("Confusion Matrix (Logistic Regression):\n", confusion_matrix(y_test, logreg_preds))

# 나이브 베이즈
gnb = GaussianNB()
gnb.fit(X_train, y_train)
gnb_preds = gnb.predict(X_test)
print("Naive Bayes 정확도:", round(accuracy_score(y_test, gnb_preds), 3))
print("Confusion Matrix (Naive Bayes):\n", confusion_matrix(y_test, gnb_preds))


Logistic Regression 정확도: 0.799
Confusion Matrix (Logistic Regression):
 [[97 13]
 [23 46]]
Naive Bayes 정확도: 0.788
Confusion Matrix (Naive Bayes):
 [[93 17]
 [21 48]]
