In [2]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression
from sklearn.tree import DecisionTreeRegressor
from sklearn.ensemble import RandomForestRegressor, GradientBoostingRegressor
from sklearn.preprocessing import StandardScaler
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score
import numpy as np

In [4]:
f_diesel = pd.read_csv(r"C:\Users\andre\OneDrive\Escritorio\USFQ\Fundamentos en Ciencia de Datos\Proyeccion_Consumo_Diesel\dataset\Consumo_Diesel_Model.csv")
f_diesel.head()

Unnamed: 0,FECHA,DISPENSADOR,HORA,VEHICULO,FLOTA,GALONES,KILOMETRAJE,ANIO,DIA_SEMANA,MES,DIA,GALONES_POR_KM,VEHICULO_ENC,FLOTA_ENC,PC1,PC2
0,2020-01-01,1,10:35:00,S024,B12M,535680.0,411475.8,2020,2,1,1,1.301851,96,2,-0.007959,0.071349
1,2020-01-01,1,12:12:00,S029,B12M,541970.0,479353.4,2020,2,1,1,1.130627,101,2,-0.012753,0.287115
2,2020-01-01,1,12:20:00,S011,B12M,565560.0,463696.8,2020,2,1,1,1.219676,83,2,-0.011493,0.237352
3,2020-01-01,1,13:10:00,S074,B12M,656140.0,451279.5,2020,2,1,1,1.453955,145,2,-0.010038,0.1979
4,2020-01-01,1,13:30:00,S015,B12M,428790.0,418781.5,2020,2,1,1,1.023899,87,2,-0.009163,0.094549


Definir características (X) y variable objetivo (y)

In [5]:
features = ['KILOMETRAJE', 'GALONES_POR_KM', 'ANIO']  # puedes incluir más
target = 'GALONES'

X = f_diesel[features]
y = f_diesel[target]

Escalado (opcional si usas modelos sensibles a la escala, como LinearRegression y GradientBoosting)

In [6]:
scaler = StandardScaler()
X_scaled = scaler.fit_transform(X)

Dividir en entrenamiento y prueba.

In [7]:
X_train, X_test, y_train, y_test = train_test_split(X_scaled, y, test_size=0.25, random_state=42)

Modelos a evaluar.

In [9]:
modelos = {
    'LinearRegression': LinearRegression(),
    'DecisionTreeRegressor': DecisionTreeRegressor(random_state=100),
    'RandomForestRegressor': RandomForestRegressor(random_state=100),
    'GradientBoostingRegressor': GradientBoostingRegressor(random_state=100),
}

Procedemos a entrenar y a evaluar

In [10]:
resultados = {}

for nombre, modelo in modelos.items():
    modelo.fit(X_train, y_train)
    y_pred = modelo.predict(X_test)
    
    resultados[nombre] = {
        'MAE': mean_absolute_error(y_test, y_pred),
        'RMSE': mean_squared_error(y_test, y_pred, squared=False),
        'R2': r2_score(y_test, y_pred)
    }




Se procede a obtener los resultados.

In [11]:
df_resultados = pd.DataFrame(resultados).T.sort_values(by='RMSE')
print("Resultados del rendimiento de los modelos:\n")
print(df_resultados)

Resultados del rendimiento de los modelos:

                                     MAE          RMSE        R2
RandomForestRegressor       50627.788493  8.831130e+06  0.963653
DecisionTreeRegressor       60404.052506  1.272420e+07  0.924543
GradientBoostingRegressor  110749.452584  1.275680e+07  0.924155
LinearRegression           545643.078395  3.896687e+07  0.292327
