In [11]:
# model_building.ipynb

import pandas as pd
import numpy as np
import os
import joblib

from sklearn.model_selection import train_test_split
from sklearn.ensemble import RandomForestRegressor
from sklearn.preprocessing import StandardScaler, OneHotEncoder
from sklearn.compose import ColumnTransformer
from sklearn.pipeline import Pipeline
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score

# 1️⃣ Load dataset
df = pd.read_csv("house_prices.csv")

# 2️⃣ Select features
numeric_features = ['OverallQual','GrLivArea','TotalBsmtSF','GarageCars','YearBuilt']
categorical_features = ['Neighborhood']
target = 'SalePrice'

# 3️⃣ Handle missing values
df[numeric_features] = df[numeric_features].fillna(df[numeric_features].median())
for col in categorical_features:
    df[col] = df[col].fillna(df[col].mode()[0])

# 4️⃣ Split X and y
X = df[numeric_features + categorical_features]
y = df[target]

X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42
)

# 5️⃣ Create preprocessing + model pipeline
preprocessor = ColumnTransformer([
    ('num', StandardScaler(), numeric_features),
    ('cat', OneHotEncoder(drop='first', handle_unknown='ignore'), categorical_features)
])

pipeline = Pipeline([
    ('preprocessor', preprocessor),
    ('model', RandomForestRegressor(random_state=42))
])

# 6️⃣ Train pipeline
pipeline.fit(X_train, y_train)

# 7️⃣ Evaluate
y_pred = pipeline.predict(X_test)
print("MAE:", mean_absolute_error(y_test, y_pred))
print("MSE:", mean_squared_error(y_test, y_pred))
print("RMSE:", np.sqrt(mean_squared_error(y_test, y_pred)))
print("R²:", r2_score(y_test, y_pred))

# 8️⃣ Save the trained pipeline
os.makedirs("model", exist_ok=True)
joblib.dump(pipeline, "model/house_price_model.pkl")

print("✅ Pipeline (model + preprocessing) saved successfully!")



MAE: 18235.665804386823
MSE: 786470961.5105069
RMSE: 28044.089600315194
R²: 0.8974657739101656
✅ Pipeline (model + preprocessing) saved successfully!
