In [1]:
from xgboost import XGBRegressor
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.ensemble import RandomForestRegressor
from sklearn.metrics import mean_squared_error
from sklearn.preprocessing import StandardScaler
from sklearn.pipeline import Pipeline
import joblib

In [2]:
# Load your dataset
df = pd.read_csv("descriptor_based_dataset.csv")  # Adjust path

In [3]:
df.head()

Unnamed: 0,MolWt,LogP,RotatableBonds,HDonors,HAcceptors,Prot_MW,Aromaticity,Instability,Hydropathy,Kd
0,320.142,3.0665,3,0,4,86942.3145,0.064935,40.691442,-0.583896,6.03292
1,235.97,1.326,1,0,0,44714.4131,0.118863,38.344496,-0.396641,4.958607
2,480.41,3.292,2,2,4,41199.1,0.119241,40.82794,0.454743,5.100015
3,480.41,3.292,2,2,4,42745.8626,0.097187,49.907698,0.501023,6.769551
4,433.723,4.6129,0,1,1,25918.2574,0.116883,36.706104,-0.198268,7.742321


In [3]:
# Feature and target selection
features = ['MolWt', 'LogP', 'RotatableBonds', 'HDonors', 'HAcceptors',
            'Prot_MW', 'Aromaticity', 'Instability', 'Hydropathy']
target = 'Kd'

X = df[features]
y = df[target]

In [4]:
# Split dataset
X_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, random_state=42)

In [5]:
xgb_pipeline = Pipeline([
    ('scaler', StandardScaler()),
    ('xgb', XGBRegressor(n_estimators=100, learning_rate=0.1, max_depth=6, verbosity=0, random_state=42))
])

In [6]:
xgb_pipeline.fit(X_train, y_train)
xgb_preds = xgb_pipeline.predict(X_val)

In [7]:
xgb_mse = mean_squared_error(y_val, xgb_preds)
print(f"Baseline XGBoost MSE: {xgb_mse:.4f}")

Baseline XGBoost MSE: 1.3551


In [8]:
joblib.dump(xgb_pipeline, "baseline_xgb_model.pkl")

['baseline_xgb_model.pkl']