In [3]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.metrics import mean_squared_error, r2_score
from catboost import CatBoostRegressor
import joblib

# Load dataset
df = pd.read_csv("Personalized_Diet_RecommendationsFE.csv")

# Define features (exclude Caloric_Intake, Recommended_Meal_Plan, Caloric_Protein)
features = [
    'Daily_Steps', 'Cholesterol_Level', 'Protein_Intake', 'Fat_Intake',
    'Carbohydrate_Intake', 'Blood_Pressure_Systolic', 'Blood_Sugar_Level',
    'Age', 'Sleep_Hours', 'Dietary_Habits_Vegan', 'Food_Aversions_Salty',
    'Macro_Average', 'Health_Score'
]
X = df[features]
y = df['Caloric_Intake']

# Split data
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Scale features
scaler = StandardScaler()
X_train = scaler.fit_transform(X_train)
X_test = scaler.transform(X_test)

# Train CatBoost Regressor
model = CatBoostRegressor(
    iterations=500,
    learning_rate=0.05,
    depth=6,
    verbose=100,
    random_state=42
)
model.fit(X_train, y_train)

# Predict and evaluate
y_pred = model.predict(X_test)
mse = mean_squared_error(y_test, y_pred)
r2 = r2_score(y_test, y_pred)
print("Regression MSE:", mse)
print("R2 Score:", r2)

# Feature Importance
feature_importance = pd.DataFrame({
    'Feature': features,
    'Importance': model.get_feature_importance()
})
feature_importance = feature_importance.sort_values(by='Importance', ascending=False)
print("\nFeature Importance:")
print(feature_importance)

# Save model and scaler
#model.save_model("catboost_regressor_no_leakage_model.cbm")
#joblib.dump(scaler, "scaler_regressor_no_leakage.pkl")
#print("\nModel saved as 'catboost_regressor_no_leakage_model.cbm'")

0:	learn: 659.1913482	total: 5.84ms	remaining: 2.91s
100:	learn: 622.9582268	total: 501ms	remaining: 1.98s
200:	learn: 590.8866097	total: 992ms	remaining: 1.48s
300:	learn: 556.9158432	total: 1.48s	remaining: 979ms
400:	learn: 526.6853305	total: 1.98s	remaining: 489ms
499:	learn: 499.3809057	total: 2.48s	remaining: 0us
Regression MSE: 454199.68573166826
R2 Score: -0.04274078982899954

Feature Importance:
                    Feature  Importance
6         Blood_Sugar_Level   11.115894
2            Protein_Intake   10.116510
1         Cholesterol_Level    9.733012
7                       Age    9.448001
8               Sleep_Hours    9.079444
3                Fat_Intake    8.407532
11            Macro_Average    8.327139
5   Blood_Pressure_Systolic    8.295643
4       Carbohydrate_Intake    7.928144
0               Daily_Steps    7.506176
12             Health_Score    7.350729
10     Food_Aversions_Salty    1.696824
9      Dietary_Habits_Vegan    0.994951
