# 1. Churn management

In [20]:
# Import necessary libraries
import pandas as pd
from sklearn.model_selection import train_test_split, cross_val_score, RandomizedSearchCV
from sklearn.ensemble import RandomForestClassifier
from sklearn.preprocessing import StandardScaler, OneHotEncoder
from sklearn.compose import ColumnTransformer
from sklearn.pipeline import Pipeline
from sklearn.metrics import classification_report, accuracy_score
import joblib

# Load the dataset into a DataFrame
df=pd.read_csv('data_cleaned.csv', encoding='ascii')

# Define the feature columns and target column
features = [
    "rev_Mean", "mou_Mean", "totmrc_Mean", "da_Mean", "ovrmou_Mean", 
    "ovrrev_Mean", "vceovr_Mean", "datovr_Mean", "drop_vce_Mean",
    "drop_blk_Mean", "ccrndmou_Mean", "plcd_vce_Mean", "comp_vce_Mean",
    "months", "uniqsubs", "actvsubs", "income", "numbcars", 
    "marital", "dwllsize",
    "prizm_social_one", "area", "dualband", "refurb_new"
]

X = df[features]
y = df['churn']

# Split the data into training and testing sets
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Identify numerical and categorical columns
numeric_features = X.select_dtypes(include=['int64', 'float64']).columns
categorical_features = X.select_dtypes(include=['object']).columns

# Data preprocessing
# Preprocessing steps for numerical and categorical columns
numeric_transformer = Pipeline(steps=[
    ('scaler', StandardScaler())
])

categorical_transformer = Pipeline(steps=[
    ('encoder', OneHotEncoder(handle_unknown='ignore', sparse_output=False))
])

# Use ColumnTransformer to apply preprocessing to the respective columns
preprocessor = ColumnTransformer(
    transformers=[
        ('num', numeric_transformer, numeric_features),
        ('cat', categorical_transformer, categorical_features)
    ]
)

# Build the pipeline: Preprocessor + RandomForestClassifier
model_pipeline = Pipeline(steps=[
    ('preprocessor', preprocessor),
    ('classifier', RandomForestClassifier(random_state=42))
])

# Step 1: Use Cross-Validation to evaluate the model
cv_scores = cross_val_score(model_pipeline, X_train, y_train, cv=5, scoring='accuracy')
print(f'Cross-Validation Accuracy: {cv_scores.mean():.4f} ± {cv_scores.std():.4f}')

# Step 2: Use RandomizedSearchCV for hyperparameter tuning
param_distributions = {
    'classifier__n_estimators': [50, 100, 200],
    'classifier__max_depth': [None, 10, 20, 30],
    'classifier__min_samples_split': [2, 5, 10],
    'classifier__min_samples_leaf': [1, 2, 4],
    'classifier__bootstrap': [True, False]
}

random_search = RandomizedSearchCV(model_pipeline, param_distributions, n_iter=100, cv=3, random_state=42, n_jobs=-1)
random_search.fit(X_train, y_train)

# Print the best parameters from RandomizedSearchCV
print(f"Best Parameters from RandomizedSearchCV: {random_search.best_params_}")

# Train the model with the best parameters
best_model = random_search.best_estimator_

# Make predictions on the test set
y_pred = best_model.predict(X_test)

# Evaluate the model
print(f"Accuracy on Test Set: {accuracy_score(y_test, y_pred):.4f}")
print("\nClassification Report:")
print(classification_report(y_test, y_pred))

# Step 3: Save the trained model to a file
# joblib.dump(best_model, 'optimized_random_forest_model.joblib')
# print("Model saved successfully.")

Cross-Validation Accuracy: 0.5875 ± 0.0028
Best Parameters from RandomizedSearchCV: {'classifier__n_estimators': 200, 'classifier__min_samples_split': 5, 'classifier__min_samples_leaf': 4, 'classifier__max_depth': 20, 'classifier__bootstrap': True}
Accuracy on Test Set: 0.6054

Classification Report:
              precision    recall  f1-score   support

           0       0.61      0.59      0.60     10021
           1       0.60      0.62      0.61      9979

    accuracy                           0.61     20000
   macro avg       0.61      0.61      0.61     20000
weighted avg       0.61      0.61      0.61     20000



In [22]:
# Import necessary libraries
import pandas as pd
from sklearn.model_selection import train_test_split, cross_val_score, RandomizedSearchCV
from sklearn.ensemble import RandomForestClassifier
from sklearn.preprocessing import StandardScaler, OneHotEncoder
from sklearn.compose import ColumnTransformer
from sklearn.pipeline import Pipeline
from sklearn.metrics import classification_report, accuracy_score
import joblib

# Load the dataset into a DataFrame
df=pd.read_csv('data_cleaned.csv', encoding='ascii')

# Define the feature columns and target column
features = [
    "mou_Mean",
    "rev_Mean",
    "totmrc_Mean",
    "roam_Mean",
    "vceovr_Mean",
    "datovr_Mean",
    "threeway_Mean",
    "peak_vce_Mean",
    "peak_dat_Mean",
    "mou_peav_Mean",
    "mou_pead_Mean",
    "eqpdays"
]

X = df[features]
y = df['churn']

# Split the data into training and testing sets
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Identify numerical and categorical columns
numeric_features = X.select_dtypes(include=['int64', 'float64']).columns
categorical_features = X.select_dtypes(include=['object']).columns

# Data preprocessing
# Preprocessing steps for numerical and categorical columns
numeric_transformer = Pipeline(steps=[
    ('scaler', StandardScaler())
])

categorical_transformer = Pipeline(steps=[
    ('encoder', OneHotEncoder(handle_unknown='ignore', sparse_output=False))
])

# Use ColumnTransformer to apply preprocessing to the respective columns
preprocessor = ColumnTransformer(
    transformers=[
        ('num', numeric_transformer, numeric_features),
        ('cat', categorical_transformer, categorical_features)
    ]
)

# Build the pipeline: Preprocessor + RandomForestClassifier
model_pipeline = Pipeline(steps=[
    ('preprocessor', preprocessor),
    ('classifier', RandomForestClassifier(random_state=42))
])

# Step 1: Use Cross-Validation to evaluate the model
cv_scores = cross_val_score(model_pipeline, X_train, y_train, cv=5, scoring='accuracy')
print(f'Cross-Validation Accuracy: {cv_scores.mean():.4f} ± {cv_scores.std():.4f}')

# Step 2: Use RandomizedSearchCV for hyperparameter tuning
param_distributions = {
    'classifier__n_estimators': [50, 100, 200],
    'classifier__max_depth': [None, 10, 20, 30],
    'classifier__min_samples_split': [2, 5, 10],
    'classifier__min_samples_leaf': [1, 2, 4],
    'classifier__bootstrap': [True, False]
}

random_search = RandomizedSearchCV(model_pipeline, param_distributions, n_iter=100, cv=3, random_state=42, n_jobs=-1)
random_search.fit(X_train, y_train)

# Print the best parameters from RandomizedSearchCV
print(f"Best Parameters from RandomizedSearchCV: {random_search.best_params_}")

# Train the model with the best parameters
best_model = random_search.best_estimator_

# Make predictions on the test set
y_pred = best_model.predict(X_test)

# Evaluate the model
print(f"Accuracy on Test Set: {accuracy_score(y_test, y_pred):.4f}")
print("\nClassification Report:")
print(classification_report(y_test, y_pred))

# Step 3: Save the trained model to a file
# joblib.dump(best_model, 'optimized_random_forest_model.joblib')
# print("Model saved successfully.")

Cross-Validation Accuracy: 0.5776 ± 0.0035
Best Parameters from RandomizedSearchCV: {'classifier__n_estimators': 200, 'classifier__min_samples_split': 5, 'classifier__min_samples_leaf': 2, 'classifier__max_depth': 10, 'classifier__bootstrap': True}
Accuracy on Test Set: 0.5951

Classification Report:
              precision    recall  f1-score   support

           0       0.61      0.54      0.57     10021
           1       0.58      0.65      0.62      9979

    accuracy                           0.60     20000
   macro avg       0.60      0.60      0.59     20000
weighted avg       0.60      0.60      0.59     20000



In [38]:
# Import necessary libraries
import pandas as pd
from sklearn.model_selection import train_test_split, cross_val_score, RandomizedSearchCV
from sklearn.ensemble import RandomForestClassifier
from sklearn.preprocessing import StandardScaler, OneHotEncoder
from sklearn.compose import ColumnTransformer
from sklearn.pipeline import Pipeline
from sklearn.metrics import classification_report, accuracy_score
import joblib

# Load the dataset into a DataFrame
df=pd.read_csv('data_cleaned.csv', encoding='ascii')

# Define the feature columns and target column
features = [
    "rev_Mean", 
    "mou_Mean", 
    "change_mou", 
    "change_rev", 
    "drop_vce_Mean", 
    "blck_vce_Mean", 
    "custcare_Mean", 
    "comp_vce_Mean", 
    "drop_blk_Mean", 
    "prizm_social_one"
]

X = df[features]
y = df['creditcd']

# Split the data into training and testing sets
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

# Identify numerical and categorical columns
numeric_features = X.select_dtypes(include=['int64', 'float64']).columns
categorical_features = X.select_dtypes(include=['object']).columns

# Data preprocessing
# Preprocessing steps for numerical and categorical columns
numeric_transformer = Pipeline(steps=[
    ('scaler', StandardScaler())
])

categorical_transformer = Pipeline(steps=[
    ('encoder', OneHotEncoder(handle_unknown='ignore', sparse_output=False))
])

# Use ColumnTransformer to apply preprocessing to the respective columns
preprocessor = ColumnTransformer(
    transformers=[
        ('num', numeric_transformer, numeric_features),
        ('cat', categorical_transformer, categorical_features)
    ]
)

# Build the pipeline: Preprocessor + RandomForestClassifier
model_pipeline = Pipeline(steps=[
    ('preprocessor', preprocessor),
    ('classifier', RandomForestClassifier(random_state=42))
])

# Step 1: Use Cross-Validation to evaluate the model
cv_scores = cross_val_score(model_pipeline, X_train, y_train, cv=5, scoring='accuracy')
print(f'Cross-Validation Accuracy: {cv_scores.mean():.4f} ± {cv_scores.std():.4f}')

# Step 2: Use RandomizedSearchCV for hyperparameter tuning
param_distributions = {
    'classifier__n_estimators': [50, 100, 200],
    'classifier__max_depth': [None, 10, 20, 30],
    'classifier__min_samples_split': [2, 5, 10],
    'classifier__min_samples_leaf': [1, 2, 4],
    'classifier__bootstrap': [True, False]
}

random_search = RandomizedSearchCV(model_pipeline, param_distributions, n_iter=100, cv=3, random_state=42, n_jobs=-1)
random_search.fit(X_train, y_train)

# Print the best parameters from RandomizedSearchCV
print(f"Best Parameters from RandomizedSearchCV: {random_search.best_params_}")

# Train the model with the best parameters
best_model = random_search.best_estimator_

# Make predictions on the test set
y_pred = best_model.predict(X_test)

# Evaluate the model
print(f"Accuracy on Test Set: {accuracy_score(y_test, y_pred):.4f}")
print("\nClassification Report:")
print(classification_report(y_test, y_pred))

# Step 3: Save the trained model to a file
# joblib.dump(best_model, 'optimized_random_forest_model.joblib')
# print("Model saved successfully.")

Cross-Validation Accuracy: 0.6801 ± 0.0016
Best Parameters from RandomizedSearchCV: {'classifier__n_estimators': 200, 'classifier__min_samples_split': 2, 'classifier__min_samples_leaf': 2, 'classifier__max_depth': 10, 'classifier__bootstrap': False}
Accuracy on Test Set: 0.6904

Classification Report:
              precision    recall  f1-score   support

           N       0.51      0.03      0.05      6197
           Y       0.69      0.99      0.81     13803

    accuracy                           0.69     20000
   macro avg       0.60      0.51      0.43     20000
weighted avg       0.64      0.69      0.58     20000

