In [4]:
import pandas as pd

In [5]:
df = pd.read_csv("../data/processed/data_11_05.csv", index_col=0)

In [6]:
y = df["price"]
X = df.drop(columns="price")

In [7]:
from sklearn.model_selection import train_test_split

X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42
)

In [9]:
import pandas as pd
import numpy as np
from sklearn.linear_model import LinearRegression
from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor
from sklearn.neighbors import KNeighborsRegressor
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score
from sklearn.preprocessing import StandardScaler

In [11]:
scaler = StandardScaler()
X_train_scaled = scaler.fit_transform(X_train)
X_test_scaled = scaler.transform(X_test)

# Инициализация моделей
models = {
    "Linear Regression": LinearRegression(),
    "Random Forest": RandomForestRegressor(random_state=42),
    "Gradient Boosting": GradientBoostingRegressor(random_state=42),
    "K-Neighbors": KNeighborsRegressor(),
}

# Обучение и оценка моделей
results = []
for name, model in models.items():
    # Обучение
    if name in ["K-Neighbors"]:
        model.fit(X_train_scaled, y_train)
        y_pred = model.predict(X_test_scaled)
    else:
        model.fit(X_train, y_train)
        y_pred = model.predict(X_test)

    # Расчет метрик
    mae = mean_absolute_error(y_test, y_pred)
    mse = mean_squared_error(y_test, y_pred)
    rmse = np.sqrt(mse)
    r2 = r2_score(y_test, y_pred)

    results.append({"Model": name, "MAE": mae, "MSE": mse, "RMSE": rmse, "R2": r2})

# Создание DataFrame с результатами
results_df = pd.DataFrame(results)
print(results_df.sort_values(by="R2", ascending=False))

               Model           MAE           MSE          RMSE        R2
1      Random Forest  1.845161e+07  2.088666e+15  4.570192e+07  0.626237
3        K-Neighbors  2.128046e+07  2.339802e+15  4.837150e+07  0.581297
0  Linear Regression  2.728058e+07  2.680125e+15  5.176992e+07  0.520396
2  Gradient Boosting  2.144657e+07  2.692167e+15  5.188609e+07  0.518242


In [12]:
from joblib import dump, load

best_model = RandomForestRegressor(random_state=42)
best_model.fit(X_train, y_train)
dump(best_model, "best_model.joblib")

['best_model.joblib']