In [1]:
import pandas as pd
from sklearn.svm import SVR
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import make_pipeline
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression
from sklearn.ensemble import RandomForestRegressor
from sklearn.metrics import mean_squared_error
import joblib

In [2]:
df = pd.read_csv("raw.csv")

# Delete the 'SLEEP TIME' and 'DATE' columns
df.drop(["SLEEP TIME", "DATE", "HEART RATE BELOW RESTING"], axis=1, inplace=True)


# Change the hours of sleep to minutes of sleep
def hours_to_minutes(time_str):
    time = time_str.split(":")
    if len(time) == 2:
        hours, minutes = time
    else:
        hours = time[0]
        minutes = time[1]
    return int(hours) * 60 + int(minutes)


df["HOURS OF SLEEP"] = df["HOURS OF SLEEP"].apply(hours_to_minutes)
df["SLEEP SCORE"] = df["SLEEP SCORE"].astype(int)

# Change percentages to range 0 - 1
for column in ["REM SLEEP", "DEEP SLEEP"]:
    df[column] = df[column].str.rstrip("%").astype(float) / 100

In [3]:
df.to_csv("cleaned.csv", index=False)

In [4]:
data = pd.read_csv("cleaned.csv")

X = data[["HOURS OF SLEEP", "REM SLEEP", "DEEP SLEEP"]]
y = data["SLEEP SCORE"]

X_train, X_test, y_train, y_test = train_test_split(
    X, y, test_size=0.2, random_state=42
)

model = LinearRegression()

model.fit(X_train, y_train)

y_pred = model.predict(X_test)

mse = mean_squared_error(y_test, y_pred)

print(f"Mean Squared Error: {mse}")

Mean Squared Error: 8.999320720820576


In [5]:
model = RandomForestRegressor(n_estimators=100, random_state=42)

model.fit(X_train, y_train)
y_pred = model.predict(X_test)

mse = mean_squared_error(y_test, y_pred)

print(f"Mean Squared Error: {mse}")

Mean Squared Error: 9.467297222222228


In [6]:
model = make_pipeline(StandardScaler(), SVR(kernel="linear"))

model.fit(X_train, y_train)
y_pred = model.predict(X_test)

mse = mean_squared_error(y_test, y_pred)

print(f"Mean Squared Error: {mse}")

Mean Squared Error: 7.635392997180669


In [7]:
joblib.dump(model, "model.joblib")

['model.joblib']

In [8]:
data

Unnamed: 0,SLEEP SCORE,HOURS OF SLEEP,REM SLEEP,DEEP SLEEP
0,63,253,0.15,0.18
1,90,489,0.22,0.18
2,83,447,0.16,0.13
3,90,445,0.19,0.21
4,81,430,0.18,0.15
...,...,...,...,...
174,83,426,0.20,0.16
175,85,417,0.20,0.18
176,91,443,0.26,0.16
177,87,468,0.22,0.21
