In [1]:
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt

In [3]:
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.linear_model import LinearRegression
from sklearn.tree import DecisionTreeRegressor
from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor
from sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error
import pickle

In [4]:
df = pd.read_csv("housing.csv")

In [5]:
df.dropna(inplace=True)
df = pd.get_dummies(df, columns=['ocean_proximity'], drop_first=True)


In [6]:
X = df.drop("median_house_value", axis=1)
y = df["median_house_value"]

In [8]:
scaler = StandardScaler()
X_scaled = scaler.fit_transform(X)

X_train, X_test, y_train, y_test = train_test_split(X_scaled, y, test_size=0.2, random_state=42)


In [10]:
models = {
    "LinearRegression": LinearRegression(),
    "DecisionTree": DecisionTreeRegressor(),
    "RandomForest": RandomForestRegressor(),
    "GradientBoosting": GradientBoostingRegressor()
}

results = {}

In [11]:
for name, model in models.items():
    model.fit(X_train, y_train)
    y_pred = model.predict(X_test)
    r2 = r2_score(y_test, y_pred)
    rmse = np.sqrt(mean_squared_error(y_test, y_pred))
    mae = mean_absolute_error(y_test, y_pred)
    results[name] = {"R2": r2, "RMSE": rmse, "MAE": mae}
    print(f"{name} => R2: {r2:.4f}, RMSE: {rmse:.2f}, MAE: {mae:.2f}")


LinearRegression => R2: 0.6488, RMSE: 69297.72, MAE: 50413.43
DecisionTree => R2: 0.6714, RMSE: 67031.71, MAE: 42878.68
RandomForest => R2: 0.8275, RMSE: 48575.04, MAE: 31518.30
GradientBoosting => R2: 0.7660, RMSE: 56564.75, MAE: 39266.16


In [12]:
best_model_name = max(results, key=lambda x: results[x]["R2"])
best_model = models[best_model_name]


In [13]:
with open("best_regression_model.pkl", "wb") as f:
    pickle.dump(best_model, f)

print(f"\n✅ Best model '{best_model_name}' saved successfully as 'best_regression_model.pkl'")


✅ Best model 'RandomForest' saved successfully as 'best_regression_model.pkl'
