In [1]:
import pandas as pd
import numpy as np
from sklearn.model_selection import train_test_split
from sklearn.ensemble import RandomForestRegressor
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score



In [2]:
df = pd.read_csv("D:\Solar Panel Project\solar_panel_dataset.csv")


In [3]:
df['Solar_Panel_Type'] = df['Solar_Panel_Type'].map({
    'Monocrystalline': 0,
    'Polycrystalline': 1,
    'Thin-film': 2
})


In [4]:
X = df[['Available_Space_m2', 'Solar_Panel_Type', 'Budget_EGP', 'Is_Sunny', 'Average_Daily_Temperature_C']]
y = df['Panels_Required']


In [5]:
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)


In [6]:
model = RandomForestRegressor(n_estimators=100, random_state=42)
model.fit(X_train, y_train)


In [7]:
y_pred = model.predict(X_test)

In [8]:
mae = mean_absolute_error(y_test, y_pred)
mse = mean_squared_error(y_test, y_pred)
r2 = r2_score(y_test, y_pred)


In [9]:
print(f"Mean Absolute Error (MAE): {mae}")
print(f"Mean Squared Error (MSE): {mse}")
print(f"R-squared (R2): {r2}")


Mean Absolute Error (MAE): 0.12269999999999998
Mean Squared Error (MSE): 0.13425050000000005
R-squared (R2): 0.9990557947196351


In [10]:
predictions_df = pd.DataFrame({
    "Actual": y_test,
    "Predicted": y_pred
})

print(predictions_df.head())

      Actual  Predicted
1860     6.0       6.78
353     33.0      33.00
1333     7.0       6.95
905     31.0      31.00
1289    39.0      39.00


In [11]:
import pickle

with open("solar_model.pkl", "wb") as f:
    pickle.dump(model, f)


In [13]:
from joblib import dump

dump(model, "solar_model.joblib")


['solar_model.joblib']