In [2]:
# Importing necessary libraries
import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns

from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression
from sklearn.tree import DecisionTreeRegressor
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score
from sklearn.preprocessing import StandardScaler, OneHotEncoder
from sklearn.compose import ColumnTransformer
from sklearn.pipeline import Pipeline

# Load the dataset (Replace with the path to your dataset)
data = pd.read_csv('/home/mitch/Downloads/Key_Economic_Indicators.csv')

# Display the first few rows to understand the structure
data.head()

# Fill missing values
data.fillna(data.median(), inplace=True)

# Feature engineering
data['AgeOfHouse'] = data['YrSold'] - data['YearBuilt']  # Age of house when sold
data['TotalBathrooms'] = data['FullBath'] + 0.5 * data['HalfBath']  # Total bathrooms
data['TotalPorchSF'] = (data['OpenPorchSF'] + data['EnclosedPorch'] +
                        data['3SsnPorch'] + data['ScreenPorch'])  # Total porch area

# Display the updated dataset with new features
data[['AgeOfHouse', 'TotalBathrooms', 'TotalPorchSF']].head()

# Selecting features and target variable
features = ['OverallQual', 'GrLivArea', 'GarageCars', 'TotalBsmtSF',
            'FullBath', 'YearBuilt', 'AgeOfHouse', 'TotalBathrooms', 'TotalPorchSF']
categorical_features = ['Neighborhood', 'BldgType']
target = 'SalePrice'

# Separate the features (X) and target variable (y)
X = data[features + categorical_features]
y = data[target]

# Split data into training and testing sets (80% for training, 20% for testing)
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Preprocessing for numeric and categorical features
numeric_features = [f for f in features if f not in categorical_features]
numeric_transformer = StandardScaler()

categorical_transformer = OneHotEncoder(handle_unknown='ignore')

# Create the ColumnTransformer
preprocessor = ColumnTransformer(
    transformers=[
        ('num', numeric_transformer, numeric_features),
        ('cat', categorical_transformer, categorical_features)
    ])

# Create a pipeline with preprocessing and linear regression
pipeline_lr = Pipeline(steps=[('preprocessor', preprocessor),
                              ('model', LinearRegression())])

# Train the model
pipeline_lr.fit(X_train, y_train)

# Make predictions
y_pred_lr = pipeline_lr.predict(X_test)

# Evaluate the Linear Regression model
print("Linear Regression Model Evaluation")
print(f"Mean Absolute Error: {mean_absolute_error(y_test, y_pred_lr):.2f}")
print(f"Mean Squared Error: {mean_squared_error(y_test, y_pred_lr):.2f}")
print(f"R^2 Score: {r2_score(y_test, y_pred_lr):.2f}")

# Create a pipeline with preprocessing and decision tree regressor
pipeline_dt = Pipeline(steps=[('preprocessor', preprocessor),
                              ('model', DecisionTreeRegressor(random_state=42))])

# Train the model
pipeline_dt.fit(X_train, y_train)

# Make predictions
y_pred_dt = pipeline_dt.predict(X_test)

# Evaluate the Decision Tree model
print("\nDecision Tree Model Evaluation")
print(f"Mean Absolute Error: {mean_absolute_error(y_test, y_pred_dt):.2f}")
print(f"Mean Squared Error: {mean_squared_error(y_test, y_pred_dt):.2f}")
print(f"R^2 Score: {r2_score(y_test, y_pred_dt):.2f}")

# Plotting Actual vs Predicted Prices for Linear Regression
plt.figure(figsize=(12, 6))
plt.scatter(y_test, y_pred_lr, color='blue', label='Linear Regression', alpha=0.6)
plt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'k--', lw=2, label="Perfect Prediction")
plt.xlabel('Actual Prices')
plt.ylabel('Predicted Prices')
plt.title('Linear Regression: Actual vs Predicted House Prices')
plt.legend()
plt.show()

# Plotting Actual vs Predicted Prices for Decision Tree
plt.figure(figsize=(12, 6))
plt.scatter(y_test, y_pred_dt, color='red', label='Decision Tree', alpha=0.6)
plt.plot([y_test.min(), y_test.max()], [y_test.min(), y_test.max()], 'k--', lw=2, label="Perfect Prediction")
plt.xlabel('Actual Prices')
plt.ylabel('Predicted Prices')
plt.title('Decision Tree: Actual vs Predicted House Prices')
plt.legend()
plt.show()


ModuleNotFoundError: No module named 'pandas'