In [2]:
import pandas as pd
from sklearn.datasets import fetch_california_housing
from sklearn.model_selection import train_test_split
from sklearn.tree import DecisionTreeRegressor
from sklearn.metrics import mean_squared_error, r2_score
import joblib


california = fetch_california_housing()
df = pd.DataFrame(california.data, columns=california.feature_names)
df['Price'] = california.target

for col in df.columns:
    df[col] = df[col].astype(str) 
    df[col] = df[col].str.replace(r"[\$,]", "", regex=True).str.strip()
    df[col] = pd.to_numeric(df[col], errors='coerce')

df = df.dropna()

X = df.drop('Price', axis=1)
y = df['Price']

X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42
)

dt_model = DecisionTreeRegressor(random_state=42)
dt_model.fit(X_train, y_train)
y_pred = dt_model.predict(X_test)

mse = mean_squared_error(y_test, y_pred)
r2 = r2_score(y_test, y_pred)

print("Decision Tree Regression Results:")
print(f"Mean Squared Error (MSE): {mse:.4f}")
print(f"R² Score: {r2:.4f}")



Decision Tree Regression Results:
Mean Squared Error (MSE): 0.4952
R² Score: 0.6221


In [4]:
import numpy as np
sample_input = np.array([[8.3252, 41, 6.984, 1.023, 322, 2.5, 37.88, -122.23]])

predicted_price = dt_model.predict(sample_input)
print(f"Predicted House Price: ${predicted_price[0]*100000:.2f}")


Predicted House Price: $382200.00


