In [13]:
# crypto_liquidity_prediction.py

import pandas as pd
import numpy as np
import matplotlib.pyplot as plt
import seaborn as sns
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler
from sklearn.ensemble import RandomForestRegressor
from sklearn.metrics import mean_absolute_error, mean_squared_error, r2_score

print("[INFO] Loading data...")
df1 = pd.read_csv("coin_gecko_2022-03-16 (1).csv")
df2 = pd.read_csv("coin_gecko_2022-03-17 (1).csv")
df = pd.concat([df1, df2], ignore_index=True)
print("[INFO] Data loaded. Shape:", df.shape)

print("[INFO] Handling missing values...")
df.fillna(df.median(numeric_only=True), inplace=True)
print("[INFO] Missing values filled.")

print("[INFO] Performing feature engineering...")
df['liquidity_ratio'] = df['24h_volume'] / df['mkt_cap']
df['volatility_7d'] = df['7d'].abs()  # proxy for volatility
print("[INFO] Feature engineering complete. Columns now:", df.columns.tolist())

features = ['price', '1h', '24h', '7d', 'volatility_7d', 'liquidity_ratio']
target = '24h_volume'

X = df[features]
y = df[target]
print("[INFO] Selected features and target.")

print("[INFO] Scaling features...")
scaler = StandardScaler()
X_scaled = scaler.fit_transform(X)
print("[INFO] Feature scaling complete.")

print("[INFO] Splitting data into train and test sets...")
X_train, X_test, y_train, y_test = train_test_split(X_scaled, y, test_size=0.2, random_state=42)
print("[INFO] Data split complete. Train shape:", X_train.shape, "Test shape:", X_test.shape)

print("[INFO] Training RandomForest model...")
model = RandomForestRegressor(n_estimators=100, random_state=42)
model.fit(X_train, y_train)
print("[INFO] Model training complete.")

print("[INFO] Making predictions...")
y_pred = model.predict(X_test)
print("[INFO] Predictions complete.")

print("[INFO] Evaluating model...")
mae = mean_absolute_error(y_test, y_pred)
rmse = np.sqrt(mean_squared_error(y_test, y_pred))
#rmse = mean_squared_error(y_test, y_pred, squared=False)

r2 = r2_score(y_test, y_pred)

print("Model Evaluation:")
print(f"MAE: {mae:.2f}")
print(f"RMSE: {rmse:.2f}")
print(f"R2 Score: {r2:.4f}")

# Optional: Save model using joblib
#import joblib
#joblib.dump(model, 'liquidity_predictor.pkl')


[INFO] Loading data...
[INFO] Data loaded. Shape: (1000, 9)
[INFO] Handling missing values...
[INFO] Missing values filled.
[INFO] Performing feature engineering...
[INFO] Feature engineering complete. Columns now: ['coin', 'symbol', 'price', '1h', '24h', '7d', '24h_volume', 'mkt_cap', 'date', 'liquidity_ratio', 'volatility_7d']
[INFO] Selected features and target.
[INFO] Scaling features...
[INFO] Feature scaling complete.
[INFO] Splitting data into train and test sets...
[INFO] Data split complete. Train shape: (800, 6) Test shape: (200, 6)
[INFO] Training RandomForest model...
[INFO] Model training complete.
[INFO] Making predictions...
[INFO] Predictions complete.
[INFO] Evaluating model...
Model Evaluation:
MAE: 233540518.51
RMSE: 832366905.98
R2 Score: -22.0628
