In [4]:
import pandas as pd
import numpy as np

from sklearn.model_selection import train_test_split
from sklearn.metrics import mean_squared_error, r2_score
from sklearn.linear_model import LinearRegression
from sklearn.ensemble import RandomForestRegressor
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import Pipeline

In [None]:
df = pd.read_csv("dataset/winequality-red.csv")

print(df.shape)
print(df.columns)
print(df.isnull().sum())


(1599, 12)
Index(['fixed acidity', 'volatile acidity', 'citric acid', 'residual sugar',
       'chlorides', 'free sulfur dioxide', 'total sulfur dioxide', 'density',
       'pH', 'sulphates', 'alcohol', 'quality'],
      dtype='object')
fixed acidity           0
volatile acidity        0
citric acid             0
residual sugar          0
chlorides               0
free sulfur dioxide     0
total sulfur dioxide    0
density                 0
pH                      0
sulphates               0
alcohol                 0
quality                 0
dtype: int64


In [6]:
print("\nTarget variable: quality")
print("\nMissing values per column:")
print(df.isnull().sum())


Target variable: quality

Missing values per column:
fixed acidity           0
volatile acidity        0
citric acid             0
residual sugar          0
chlorides               0
free sulfur dioxide     0
total sulfur dioxide    0
density                 0
pH                      0
sulphates               0
alcohol                 0
quality                 0
dtype: int64


In [7]:
X = df.drop(columns=["quality"])
y = df["quality"]

X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42
)

print("Train/Test split completed.")
print("Training samples:", X_train.shape[0])
print("Testing samples:", X_test.shape[0])


Train/Test split completed.
Training samples: 1279
Testing samples: 320


In [None]:
pipe = Pipeline([
    ("scaler", StandardScaler()),
    ("lr", LinearRegression())
])

pipe.fit(X_train, y_train)

pred2 = pipe.predict(X_test)
mse2 = mean_squared_error(y_test, pred2)
r22 = r2_score(y_test, pred2)

print("EXP-02: Linear Regression + StandardScaler")
print("MSE:", mse2)
print("R² Score:", r22)


EXP-01: Linear Regression (No Preprocessing)
MSE: 0.3900251439643167
R² Score: 0.4031803412790683


In [None]:
import os
import json

metrics_exp2 = {
    "model": "Linear Regression + StandardScaler",
    "mse": mse2,
    "r2_score": r22
}

os.makedirs("outputs/results", exist_ok=True)

with open("outputs/results/metrics.json", "w") as f:
    json.dump(metrics_exp2, f, indent=4)

print("Metrics for EXP-02 saved successfully!")
