In [None]:
import pandas as pd
import numpy as np

from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import Pipeline

import joblib


In [None]:
df = pd.read_csv("train.csv")  # Kaggle House Prices dataset


In [None]:
features = [
    "OverallQual",
    "GrLivArea",
    "TotalBsmtSF",
    "GarageCars",
    "FullBath",
    "YearBuilt"
]

X = df[features]
y = df["SalePrice"]


In [None]:
X = X.fillna(X.median())


In [None]:
X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42
)


In [None]:
model = Pipeline([
    ("scaler", StandardScaler()),
    ("regressor", LinearRegression())
])


In [None]:
model.fit(X_train, y_train)


In [None]:
y_pred = model.predict(X_test)

mae = mean_absolute_error(y_test, y_pred)
mse = mean_squared_error(y_test, y_pred)
rmse = np.sqrt(mse)
r2 = r2_score(y_test, y_pred)

print("MAE:", mae)
print("MSE:", mse)
print("RMSE:", rmse)
print("RÂ² Score:", r2)


In [None]:
joblib.dump(model, "house_price_model.pkl")
print("Model saved successfully!")


In [None]:
loaded_model = joblib.load("house_price_model.pkl")
