In [12]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.linear_model import LinearRegression
from sklearn.metrics import mean_squared_error
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import Pipeline
import joblib

In [13]:
# Load your dataset
df = pd.read_csv("descriptor_based_dataset.csv")  # Adjust path

In [14]:
df.isna().sum()

MolWt             0
LogP              0
RotatableBonds    0
HDonors           0
HAcceptors        0
Prot_MW           0
Aromaticity       0
Instability       0
Hydropathy        0
Kd                0
dtype: int64

In [15]:
# Feature and target selection
features = ['MolWt', 'LogP', 'RotatableBonds', 'HDonors', 'HAcceptors',
            'Prot_MW', 'Aromaticity', 'Instability', 'Hydropathy']
target = 'Kd'

X = df[features]
y = df[target]

In [16]:
# Split dataset
X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, random_state=42)

In [17]:
linear_pipeline = Pipeline([
    ('scaler', StandardScaler()),
    ('lr', LinearRegression())
])

In [18]:
linear_pipeline.fit(X_train, y_train)
lr_preds = linear_pipeline.predict(X_val)

In [19]:
lr_mse = mean_squared_error(y_val, lr_preds)
print(f"Baseline Linear Regression MSE: {lr_mse:.4f}")

Baseline Linear Regression MSE: 2.3159


In [20]:
joblib.dump(linear_pipeline, "baseline_linear_model.pkl")

['baseline_linear_model.pkl']