In [None]:
!pip install fosforml 
!pip install fosforio

In [1]:
import pandas as pd
from sklearn.model_selection import train_test_split
from sklearn.preprocessing import StandardScaler, OneHotEncoder
from sklearn.compose import ColumnTransformer
from sklearn.pipeline import Pipeline
from sklearn.linear_model import LogisticRegression
from sklearn.metrics import classification_report

In [2]:
from fosforml.model_manager.snowflakesession import get_session
my_session = get_session()

In [3]:
df = 'EMPLOYEE_DATASET'

In [4]:
sf_df = my_session.sql("select * from {}".format(df))

In [5]:
df = sf_df.to_pandas()

In [6]:
# Preprocessing
numeric_features = ['AGE', 'SALARY_INR', 'AVERAGE_PERCENTAGE_SALARY_HIKE', 'AVERAGE_PERFORMANCE_RATING', 'OVERTIME_HOURS', 'TENURE_MONTHS']
categorical_features = ['QUALIFICATION', 'ETHNICITY', 'MARITAL_STATUS', 'GENDER', 'CONTINENT', 'COUNTRY', 'STATE', 'CITY', 'ROLE', 'LINE_OF_BUSINESS', 'DELIVERY_UNIT', 'PRACTICE_UNIT', 'EMPLOYMENT_TYPE', 'TURNOVER_REASONS', 'SHIFT', 'OVER_TIME']

In [7]:
preprocessor = ColumnTransformer(
    transformers=[
        ('num', StandardScaler(), numeric_features),
        ('cat', OneHotEncoder(), categorical_features)])

In [9]:
# Split data
X = df.drop('CHURN', axis=1)
y = df['CHURN']
X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)

In [10]:
# Create pipeline
clf = Pipeline(steps=[('preprocessor', preprocessor),
                      ('classifier', LogisticRegression())])

In [11]:
# Train model
clf.fit(X_train, y_train)

In [12]:
# Predict and evaluate
y_pred = clf.predict(X_test)
print(classification_report(y_test, y_pred))

              precision    recall  f1-score   support

           0       0.85      0.72      0.78     29990
           1       0.76      0.87      0.81     30010

    accuracy                           0.80     60000
   macro avg       0.80      0.80      0.80     60000
weighted avg       0.80      0.80      0.80     60000

