In [2]:
#Import the relevant libraries
import numpy as np
import pandas as pd
import statsmodels.api as sm
import matplotlib.pyplot as plt
import seaborn as sns
sns.set()

In [3]:
#Read and save the data
raw_data = pd.read_csv('2.02.+Binary+predictors.csv')
raw_data.head()

Unnamed: 0,SAT,Admitted,Gender
0,1363,No,Male
1,1792,Yes,Female
2,1954,Yes,Female
3,1653,No,Male
4,1593,No,Male


In [4]:
#Create dummies for Gender
data = raw_data.copy()
data['Admitted'] = data['Admitted'].map({'Yes':1, 'No':0})
data['Gender'] = data['Gender'].map({'Female':1, 'Male':0})
data.head()
#In here the reference will be male and the comparison group will be Female, that's why male is 0 and female is 1.

Unnamed: 0,SAT,Admitted,Gender
0,1363,0,0
1,1792,1,1
2,1954,1,1
3,1653,0,0
4,1593,0,0


In [5]:
#Create the independent and dependent variables
x1 = data[['SAT', 'Gender']]
y = data['Admitted']

In [6]:
#Create the regression
x = sm.add_constant(x1)
reg_log = sm.Logit(y,x)
log_result = reg_log.fit()

Optimization terminated successfully.
         Current function value: 0.120117
         Iterations 10


In [7]:
#Display the summary
log_result.summary()
#For every 1 unit increase in SAT the odds increase by 4.2%
#Women are 7x more likely to be admitted than men.

0,1,2,3
Dep. Variable:,Admitted,No. Observations:,168.0
Model:,Logit,Df Residuals:,165.0
Method:,MLE,Df Model:,2.0
Date:,"Mon, 01 Dec 2025",Pseudo R-squ.:,0.8249
Time:,18:28:20,Log-Likelihood:,-20.18
converged:,True,LL-Null:,-115.26
Covariance Type:,nonrobust,LLR p-value:,5.1180000000000006e-42

0,1,2,3,4,5,6
,coef,std err,z,P>|z|,[0.025,0.975]
const,-68.3489,16.454,-4.154,0.000,-100.598,-36.100
SAT,0.0406,0.010,4.129,0.000,0.021,0.060
Gender,1.9449,0.846,2.299,0.022,0.287,3.603


In [8]:
#Accuracy evaluation
np.set_printoptions(formatter={'float': lambda x: "{0:0.2f}".format(x)})
log_result.predict()
#Predicted values

array([0.00, 1.00, 1.00, 0.23, 0.02, 0.99, 1.00, 1.00, 1.00, 0.01, 1.00,
       1.00, 0.76, 0.00, 0.60, 1.00, 0.11, 0.12, 0.51, 1.00, 1.00, 1.00,
       0.00, 0.01, 0.97, 1.00, 0.48, 0.99, 1.00, 0.99, 0.00, 0.83, 0.25,
       1.00, 1.00, 1.00, 0.31, 1.00, 0.23, 0.00, 0.02, 0.45, 1.00, 0.00,
       0.99, 0.00, 0.99, 0.00, 0.00, 0.01, 0.00, 1.00, 0.92, 0.02, 1.00,
       0.00, 0.37, 0.98, 0.12, 1.00, 0.00, 0.78, 1.00, 1.00, 0.98, 0.00,
       0.00, 0.00, 1.00, 0.00, 0.78, 0.12, 0.00, 0.99, 1.00, 1.00, 0.00,
       0.30, 1.00, 1.00, 0.00, 1.00, 1.00, 0.85, 1.00, 1.00, 0.00, 1.00,
       1.00, 0.89, 0.83, 0.00, 0.98, 0.97, 0.00, 1.00, 1.00, 0.03, 0.99,
       0.96, 1.00, 0.00, 1.00, 0.01, 0.01, 1.00, 1.00, 1.00, 0.00, 0.00,
       0.02, 0.33, 0.00, 1.00, 0.09, 0.00, 0.97, 0.00, 0.75, 1.00, 1.00,
       0.01, 0.01, 0.00, 1.00, 0.00, 0.99, 0.57, 0.54, 0.87, 0.83, 0.00,
       1.00, 0.00, 0.00, 0.00, 1.00, 0.04, 0.00, 0.01, 1.00, 0.99, 0.52,
       1.00, 1.00, 0.05, 0.00, 0.00, 0.00, 0.68, 1.

In [11]:
#Actual values
np.array(data['Admitted'])

array([0, 1, 1, 0, 0, 1, 1, 1, 1, 0, 1, 1, 1, 0, 0, 1, 0, 0, 1, 1, 1, 1,
       0, 0, 1, 1, 1, 1, 1, 1, 0, 1, 0, 1, 1, 1, 0, 1, 0, 0, 0, 1, 1, 0,
       1, 0, 1, 0, 0, 0, 0, 1, 0, 0, 1, 0, 0, 1, 0, 1, 0, 1, 1, 1, 1, 0,
       0, 0, 1, 0, 1, 1, 0, 1, 1, 1, 0, 1, 1, 1, 0, 1, 1, 0, 1, 1, 0, 1,
       1, 1, 0, 0, 1, 1, 0, 1, 1, 0, 1, 1, 1, 0, 1, 0, 0, 1, 1, 1, 0, 0,
       0, 0, 0, 1, 0, 0, 1, 0, 1, 1, 1, 0, 0, 0, 1, 0, 1, 0, 1, 1, 1, 0,
       1, 0, 0, 0, 1, 0, 0, 0, 1, 1, 1, 1, 1, 0, 0, 0, 0, 1, 1, 1, 1, 1,
       1, 0, 1, 1, 0, 1, 0, 1, 1, 1, 1, 0, 0, 0], dtype=int64)

In [15]:
#Table that compares the predicted values against the actual values (values below 50% are 0 and values above 50% are 1)
log_result.pred_table()

array([[69.00, 5.00],
       [4.00, 90.00]])

In [19]:
cm_df = pd.DataFrame(log_result.pred_table())
cm_df.columns = ['Predicted 0', 'Predicted 1']
cm_df = cm_df.rename(index = {0: 'Actual 0',1:'Actual 1'})
cm_df

Unnamed: 0,Predicted 0,Predicted 1
Actual 0,69.0,5.0
Actual 1,4.0,90.0


In [30]:
#Calculating the accuracy
cm = np.array(cm_df)
accuracy_train = (cm[0,0]+cm[1,1])/cm.sum()
accuracy_train
#Our model has a 95% of accuracy (159 predictions out of 168 were correct)

0.9464285714285714

In [39]:
#Testing the model

#Read the data
test = pd.read_csv('2.03.+Test+dataset.csv')
test.head()

Unnamed: 0,SAT,Admitted,Gender
0,1323,No,Male
1,1725,Yes,Female
2,1762,Yes,Female
3,1777,Yes,Male
4,1665,No,Male


In [40]:
#Map the data
test['Admitted'] = test['Admitted'].map({'Yes':1, 'No':0})
test['Gender'] = test['Gender'].map({'Male':1, 'Female':0})
test

Unnamed: 0,SAT,Admitted,Gender
0,1323,0,1
1,1725,1,0
2,1762,1,0
3,1777,1,1
4,1665,0,1
5,1556,1,0
6,1731,1,0
7,1809,1,0
8,1930,1,0
9,1708,1,1


In [42]:
x

Unnamed: 0,const,SAT,Gender
0,1.0,1363,0
1,1.0,1792,1
2,1.0,1954,1
3,1.0,1653,0
4,1.0,1593,0
...,...,...,...
163,1.0,1722,1
164,1.0,1750,0
165,1.0,1555,0
166,1.0,1524,0


In [44]:
test_actual = test['Admitted']
test_data = test.drop(['Admitted'],axis=1)

In [46]:
test_data = sm.add_constant(test_data)
test_data

Unnamed: 0,const,SAT,Gender
0,1.0,1323,1
1,1.0,1725,0
2,1.0,1762,0
3,1.0,1777,1
4,1.0,1665,1
5,1.0,1556,0
6,1.0,1731,0
7,1.0,1809,0
8,1.0,1930,0
9,1.0,1708,1


In [47]:
#Function to predict the test data using the confusion matrix
def confusion_matrix(data,actual_values,model):
    
    pred_values = model.predict(data)
    bins = np.array([0,0.5,1])
    cm = np.histogram2d(actual_values, pred_values, bins=bins)[0]
    accuracy = (cm[0,0]+cm[1,1])/cm.sum()
    return cm, accuracy

In [49]:
cm = confusion_matrix(test_data,test_actual,log_result)
cm

(array([[4.00, 2.00],
        [2.00, 11.00]]),
 0.7894736842105263)

In [51]:
cm_df = pd.DataFrame(cm[0])
cm_df.columns = ['Predicted 0', 'Predicted 1']
cm_df = cm_df.rename(index={0: 'Actual 0',1:'Actual 1'})
cm_df

Unnamed: 0,Predicted 0,Predicted 1
Actual 0,4.0,2.0
Actual 1,2.0,11.0


In [52]:
#Print the missclassifaction rate from the model -- if multiplied by 100 it gives us a percentage of how much of the data our model did not predicted well.
print('Missclassification rate: '+str((1+1)/19))

Missclassification rate: 0.10526315789473684
