In [1]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import LabelEncoder, StandardScaler
from sklearn.ensemble import RandomForestClassifier
from sklearn.metrics import accuracy_score

In [2]:
train_path = 'train.csv'
test_path = 'test.csv'

In [3]:
train_data = pd.read_csv(train_path)
test_data = pd.read_csv(test_path)


In [4]:
X = train_data.drop(columns=['y'])
y = train_data['y']

In [5]:
categorical_columns = X.select_dtypes(include=['object']).columns
numerical_columns = X.select_dtypes(include=['int64']).columns


In [6]:
train_data.info()

<class 'pandas.core.frame.DataFrame'>
RangeIndex: 40000 entries, 0 to 39999
Data columns (total 17 columns):
 #   Column     Non-Null Count  Dtype 
---  ------     --------------  ----- 
 0   age        40000 non-null  int64 
 1   job        40000 non-null  object
 2   marital    40000 non-null  object
 3   education  40000 non-null  object
 4   default    40000 non-null  object
 5   balance    40000 non-null  int64 
 6   housing    40000 non-null  object
 7   loan       40000 non-null  object
 8   contact    40000 non-null  object
 9   day        40000 non-null  int64 
 10  month      40000 non-null  object
 11  duration   40000 non-null  int64 
 12  campaign   40000 non-null  int64 
 13  pdays      40000 non-null  int64 
 14  previous   40000 non-null  int64 
 15  poutcome   40000 non-null  object
 16  y          40000 non-null  object
dtypes: int64(7), object(10)
memory usage: 5.2+ MB


In [7]:
label_encoder = LabelEncoder()
for col in categorical_columns:
    # Combine train and test to handle unseen labels
    combined_data = pd.concat([X[col], test_data[col]], axis=0).astype(str)
    label_encoder.fit(combined_data)
    X[col] = label_encoder.transform(X[col].astype(str))
    test_data[col] = label_encoder.transform(test_data[col].astype(str))


In [8]:
scaler = StandardScaler()
X[numerical_columns] = scaler.fit_transform(X[numerical_columns])
test_data[numerical_columns] = scaler.transform(test_data[numerical_columns])

In [9]:
X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.2, random_state=42)


In [10]:
model = RandomForestClassifier(random_state=42)
model.fit(X_train, y_train)

In [11]:
y_val_pred = model.predict(X_val)
val_accuracy = accuracy_score(y_val, y_val_pred)
print(f"Validation Accuracy: {val_accuracy:.2f}")

Validation Accuracy: 0.94


In [12]:
test_data['y'] = model.predict(test_data)

In [14]:
output_path = 'test.csv'
test_data.to_csv(output_path, index=False)
print(f"Predictions saved to {output_path}")

Predictions saved to test.csv
