In [1]:
import pandas as pd
import numpy as np
import time
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from tensorflow.keras.models import Sequential
from tensorflow.keras.layers import Dense
from tensorflow.keras.optimizers import Adam
from sklearn.metrics import mean_squared_error

In [3]:
df = pd.read_csv("sampled_dfdata.csv")

df = df.dropna()
X = df.drop("outcome", axis=1)
y = df["outcome"]

# Standardize features
scaler = StandardScaler()
X_scaled = scaler.fit_transform(X)

# Function to train and evaluate model
def train_model(X, y, hidden_layers=(4,), sample_size=1000, epochs=20):
    # Take subset
    X_sub, _, y_sub, _ = train_test_split(X, y, train_size=sample_size, stratify=y, random_state=42)
    # Split into train and validation
    X_train, X_val, y_train, y_val = train_test_split(X_sub, y_sub, test_size=0.2, random_state=42)

    # Build model
    model = Sequential()
    model.add(Dense(hidden_layers[0], input_shape=(X.shape[1],), activation='relu'))
    for nodes in hidden_layers[1:]:
        model.add(Dense(nodes, activation='relu'))
    model.add(Dense(1, activation='sigmoid'))

    model.compile(optimizer=Adam(learning_rate=0.001), loss='binary_crossentropy')

    # Train
    start = time.time()
    history = model.fit(X_train, y_train, epochs=epochs, batch_size=32, verbose=0)
    end = time.time()

    # Evaluate
    y_train_pred = model.predict(X_train).flatten()
    y_val_pred = model.predict(X_val).flatten()
    train_error = mean_squared_error(y_train, y_train_pred)
    val_error = mean_squared_error(y_val, y_val_pred)

    return train_error, val_error, end - start

# Configurations
configs = [
    (1000, (4,)),
    (10000, (4,)),
    (100000, (4,)),
    (1000, (4, 4)),
    (10000, (4, 4)),
    (100000, (4, 4)),
]

# Run experiments
results = []
for sample_size, layers in configs:
    # Limit sample size to available data
    size = min(sample_size, len(X))
    train_err, val_err, exec_time = train_model(X_scaled, y, hidden_layers=layers, sample_size=size)
    results.append((size, layers, train_err, val_err, exec_time))

# Show results
for res in results:
    print(f"Data size: {res[0]}, Layers: {res[1]}, Train Error: {res[2]:.4f}, Val Error: {res[3]:.4f}, Time: {res[4]:.2f}s")

  super().__init__(activity_regularizer=activity_regularizer, **kwargs)


[1m25/25[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m0s[0m 2ms/step 
[1m7/7[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m0s[0m 8ms/step 


  super().__init__(activity_regularizer=activity_regularizer, **kwargs)


[1m250/250[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m0s[0m 1ms/step  
[1m63/63[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m0s[0m 2ms/step


  super().__init__(activity_regularizer=activity_regularizer, **kwargs)


[1m2500/2500[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m3s[0m 987us/step
[1m625/625[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m1s[0m 1ms/step


  super().__init__(activity_regularizer=activity_regularizer, **kwargs)


[1m25/25[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m0s[0m 2ms/step 
[1m7/7[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m0s[0m 9ms/step 


  super().__init__(activity_regularizer=activity_regularizer, **kwargs)


[1m250/250[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m0s[0m 2ms/step
[1m63/63[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m0s[0m 2ms/step


  super().__init__(activity_regularizer=activity_regularizer, **kwargs)


[1m2500/2500[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m3s[0m 1ms/step
[1m625/625[0m [32m━━━━━━━━━━━━━━━━━━━━[0m[37m[0m [1m1s[0m 1ms/step
Data size: 1000, Layers: (4,), Train Error: 0.1441, Val Error: 0.1332, Time: 3.34s
Data size: 10000, Layers: (4,), Train Error: 0.0068, Val Error: 0.0071, Time: 11.34s
Data size: 100000, Layers: (4,), Train Error: 0.0020, Val Error: 0.0019, Time: 95.26s
Data size: 1000, Layers: (4, 4), Train Error: 0.1610, Val Error: 0.1457, Time: 3.13s
Data size: 10000, Layers: (4, 4), Train Error: 0.0034, Val Error: 0.0041, Time: 13.34s
Data size: 100000, Layers: (4, 4), Train Error: 0.0014, Val Error: 0.0014, Time: 97.24s
