## Importing Libraries

In [123]:
import numpy as np
import pandas as pd
import matplotlib.pyplot as plt
import seaborn as sns
from matplotlib.ticker import PercentFormatter
sns.set()

## Importing the Dataset uding numpy

In [124]:
Absentsent = pd.read_csv('absenteeism-data.csv')

In [125]:
# View the first five records from the dataset
Absentsent.head()

Unnamed: 0,ID,Reason for Absence,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours
0,11,26,07/07/2015,289,36,33,239.554,30,1,2,1,4
1,36,0,14/07/2015,118,13,50,239.554,31,1,1,0,0
2,3,23,15/07/2015,179,51,38,239.554,31,1,0,0,2
3,7,7,16/07/2015,279,5,39,239.554,24,1,2,0,4
4,11,23,23/07/2015,289,36,33,239.554,30,1,2,1,2


In [126]:
# Copying the dataset
df = Absentsent.copy()

In [128]:
pd.options.display.max_columns = None
pd.options.display.max_rows = None

In [129]:
df.head()

Unnamed: 0,ID,Reason for Absence,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours
0,11,26,07/07/2015,289,36,33,239.554,30,1,2,1,4
1,36,0,14/07/2015,118,13,50,239.554,31,1,1,0,0
2,3,23,15/07/2015,179,51,38,239.554,31,1,0,0,2
3,7,7,16/07/2015,279,5,39,239.554,24,1,2,0,4
4,11,23,23/07/2015,289,36,33,239.554,30,1,2,1,2


## Statistical Analysis

In [130]:
df.info()

<class 'pandas.core.frame.DataFrame'>
RangeIndex: 700 entries, 0 to 699
Data columns (total 12 columns):
 #   Column                     Non-Null Count  Dtype  
---  ------                     --------------  -----  
 0   ID                         700 non-null    int64  
 1   Reason for Absence         700 non-null    int64  
 2   Date                       700 non-null    object 
 3   Transportation Expense     700 non-null    int64  
 4   Distance to Work           700 non-null    int64  
 5   Age                        700 non-null    int64  
 6   Daily Work Load Average    700 non-null    float64
 7   Body Mass Index            700 non-null    int64  
 8   Education                  700 non-null    int64  
 9   Children                   700 non-null    int64  
 10  Pets                       700 non-null    int64  
 11  Absenteeism Time in Hours  700 non-null    int64  
dtypes: float64(1), int64(10), object(1)
memory usage: 65.8+ KB


In [131]:
df.describe()

Unnamed: 0,ID,Reason for Absence,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours
count,700.0,700.0,700.0,700.0,700.0,700.0,700.0,700.0,700.0,700.0,700.0
mean,17.951429,19.411429,222.347143,29.892857,36.417143,271.801774,26.737143,1.282857,1.021429,0.687143,6.761429
std,11.028144,8.356292,66.31296,14.804446,6.379083,40.021804,4.254701,0.66809,1.112215,1.166095,12.670082
min,1.0,0.0,118.0,5.0,27.0,205.917,19.0,1.0,0.0,0.0,0.0
25%,9.0,13.0,179.0,16.0,31.0,241.476,24.0,1.0,0.0,0.0,2.0
50%,18.0,23.0,225.0,26.0,37.0,264.249,25.0,1.0,1.0,0.0,3.0
75%,28.0,27.0,260.0,50.0,40.0,294.217,31.0,1.0,2.0,1.0,8.0
max,36.0,28.0,388.0,52.0,58.0,378.884,38.0,4.0,4.0,8.0,120.0


# Transforming the Dataset

## Drop ID

In [134]:
# Dropping the ID column because it won't contribute any insights
df = df.drop('ID', axis = 1)

KeyError: "['ID'] not found in axis"

In [135]:
df.head()

Unnamed: 0,Reason for Absence,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours
0,26,07/07/2015,289,36,33,239.554,30,1,2,1,4
1,0,14/07/2015,118,13,50,239.554,31,1,1,0,0
2,23,15/07/2015,179,51,38,239.554,31,1,0,0,2
3,7,16/07/2015,279,5,39,239.554,24,1,2,0,4
4,23,23/07/2015,289,36,33,239.554,30,1,2,1,2


## Reason for Absence

In [136]:
df['Reason for Absence'].min()

0

In [137]:
df['Reason for Absence'].max()

28

In [138]:
df['Reason for Absence'].unique()

array([26,  0, 23,  7, 22, 19,  1, 11, 14, 21, 10, 13, 28, 18, 25, 24,  6,
       27, 17,  8, 12,  5,  9, 15,  4,  3,  2, 16], dtype=int64)

In [139]:
sorted(df['Reason for Absence'].unique())

[0,
 1,
 2,
 3,
 4,
 5,
 6,
 7,
 8,
 9,
 10,
 11,
 12,
 13,
 14,
 15,
 16,
 17,
 18,
 19,
 21,
 22,
 23,
 24,
 25,
 26,
 27,
 28]

In [141]:
# Use get_dummies function to create dummy variables for reasons for absence column
reason_columns = pd.get_dummies(df['Reason for Absence'])

In [142]:
reason_columns.head()

Unnamed: 0,0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,21,22,23,24,25,26,27,28
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0
1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0
2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0
3,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0
4,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0


In [143]:
reason_columns['check'] = reason_columns.sum(axis=1)
reason_columns.head()

Unnamed: 0,0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,21,22,23,24,25,26,27,28,check
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,1
1,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1
2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,1
3,0,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1
4,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0,1


In [144]:
reason_columns['check'].sum(axis = 0)

700

In [145]:
reason_columns['check'].unique()

array([1], dtype=int64)

In [146]:
reason_columns = reason_columns.drop(['check'], axis = 1)

In [148]:
reason_columns = pd.get_dummies(df['Reason for Absence'], drop_first = True)

In [149]:
reason_columns.head()

Unnamed: 0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,21,22,23,24,25,26,27,28
0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0
1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0
2,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0
3,0,0,0,0,0,0,1,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0
4,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,0,1,0,0,0,0,0


In [150]:
df.columns.values

array(['Reason for Absence', 'Date', 'Transportation Expense',
       'Distance to Work', 'Age', 'Daily Work Load Average',
       'Body Mass Index', 'Education', 'Children', 'Pets',
       'Absenteeism Time in Hours'], dtype=object)

In [151]:
reason_columns.columns.values

array([ 1,  2,  3,  4,  5,  6,  7,  8,  9, 10, 11, 12, 13, 14, 15, 16, 17,
       18, 19, 21, 22, 23, 24, 25, 26, 27, 28], dtype=int64)

In [152]:
df = df.drop('Reason for Absence', axis = 1)

In [153]:
df.head()

Unnamed: 0,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours
0,07/07/2015,289,36,33,239.554,30,1,2,1,4
1,14/07/2015,118,13,50,239.554,31,1,1,0,0
2,15/07/2015,179,51,38,239.554,31,1,0,0,2
3,16/07/2015,279,5,39,239.554,24,1,2,0,4
4,23/07/2015,289,36,33,239.554,30,1,2,1,2


In [154]:
# categories the reasons for absent into groups
reason_1 = reason_columns.loc[:, 1:14].max(axis = 1)
reason_2 = reason_columns.loc[:, 15:17].max(axis = 1)
reason_3 = reason_columns.loc[:, 18:21].max(axis = 1)
reason_4 = reason_columns.loc[:, 22:].max(axis = 1)

In [155]:
df = pd.concat([df,reason_1,reason_2,reason_3,reason_4], axis = 1)

In [156]:
df.head()

Unnamed: 0,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours,0,1,2,3
0,07/07/2015,289,36,33,239.554,30,1,2,1,4,0,0,0,1
1,14/07/2015,118,13,50,239.554,31,1,1,0,0,0,0,0,0
2,15/07/2015,179,51,38,239.554,31,1,0,0,2,0,0,0,1
3,16/07/2015,279,5,39,239.554,24,1,2,0,4,1,0,0,0
4,23/07/2015,289,36,33,239.554,30,1,2,1,2,0,0,0,1


In [157]:
column_names = ['Date', 'Transportation Expense', 'Distance to Work', 'Age',
       'Daily Work Load Average', 'Body Mass Index', 'Education',
       'Children', 'Pets', 'Absenteeism Time in Hours', 'Reason_1', 'Reason_2', 'Reason_3', 'Reason_4']

In [158]:
df.columns = column_names

In [159]:
df.head()

Unnamed: 0,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours,Reason_1,Reason_2,Reason_3,Reason_4
0,07/07/2015,289,36,33,239.554,30,1,2,1,4,0,0,0,1
1,14/07/2015,118,13,50,239.554,31,1,1,0,0,0,0,0,0
2,15/07/2015,179,51,38,239.554,31,1,0,0,2,0,0,0,1
3,16/07/2015,279,5,39,239.554,24,1,2,0,4,1,0,0,0
4,23/07/2015,289,36,33,239.554,30,1,2,1,2,0,0,0,1


In [160]:
column_names_reordered = ['Reason_1', 'Reason_2', 'Reason_3', 'Reason_4', 
                          'Date', 'Transportation Expense', 'Distance to Work', 'Age',
       'Daily Work Load Average', 'Body Mass Index', 'Education',
       'Children', 'Pets', 'Absenteeism Time in Hours']

In [161]:
df = df[column_names_reordered]

In [162]:
df.head()

Unnamed: 0,Reason_1,Reason_2,Reason_3,Reason_4,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours
0,0,0,0,1,07/07/2015,289,36,33,239.554,30,1,2,1,4
1,0,0,0,0,14/07/2015,118,13,50,239.554,31,1,1,0,0
2,0,0,0,1,15/07/2015,179,51,38,239.554,31,1,0,0,2
3,1,0,0,0,16/07/2015,279,5,39,239.554,24,1,2,0,4
4,0,0,0,1,23/07/2015,289,36,33,239.554,30,1,2,1,2


In [225]:
reasons = df.copy()

In [226]:
reasons.head()

Unnamed: 0,Reason_1,Reason_2,Reason_3,Reason_4,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours
0,0,0,0,1,07/07/2015,289,36,33,239.554,30,1,2,1,4
1,0,0,0,0,14/07/2015,118,13,50,239.554,31,1,1,0,0
2,0,0,0,1,15/07/2015,179,51,38,239.554,31,1,0,0,2
3,1,0,0,0,16/07/2015,279,5,39,239.554,24,1,2,0,4
4,0,0,0,1,23/07/2015,289,36,33,239.554,30,1,2,1,2


## Date

In [227]:
type(reasons['Date'][0])

str

In [228]:
# Convert the date column to a datetime format
reasons['Date'] = pd.to_datetime(reasons['Date'])

In [229]:
reasons['Date'].head()

0   2015-07-07
1   2015-07-14
2   2015-07-15
3   2015-07-16
4   2015-07-23
Name: Date, dtype: datetime64[ns]

In [199]:
reasons.info()

<class 'pandas.core.frame.DataFrame'>
RangeIndex: 700 entries, 0 to 699
Data columns (total 14 columns):
 #   Column                     Non-Null Count  Dtype         
---  ------                     --------------  -----         
 0   Reason_1                   700 non-null    uint8         
 1   Reason_2                   700 non-null    uint8         
 2   Reason_3                   700 non-null    uint8         
 3   Reason_4                   700 non-null    uint8         
 4   Date                       700 non-null    datetime64[ns]
 5   Transportation Expense     700 non-null    int64         
 6   Distance to Work           700 non-null    int64         
 7   Age                        700 non-null    int64         
 8   Daily Work Load Average    700 non-null    float64       
 9   Body Mass Index            700 non-null    int64         
 10  Education                  700 non-null    int64         
 11  Children                   700 non-null    int64         
 12  Pets    

In [200]:
reasons['Date'][0].month

7

In [230]:
list_months = []
list_months

[]

In [231]:
for i in range(reasons.shape[0]):
    list_months.append(reasons['Date'][i].month)

In [232]:
list_months[:5]

[7, 7, 7, 7, 7]

In [233]:
reasons['Month'] = list_months

In [234]:
reasons['Month'] = reasons['Month'].map({1: 'January', 2: 'February', 3: 'March', 4: 'April', 5: 'May'
                                        ,6: 'June', 7: 'July', 8: 'August', 9: 'September', 10: 'October', 
                                         11: 'November', 12: 'December'})
reasons['Month'].head()

0    July
1    July
2    July
3    July
4    July
Name: Month, dtype: object

In [206]:
reasons['Date'][699].day_name()

'Thursday'

In [235]:
Week_day_name, Week_day_numeric = [], []

In [236]:
for i in range(reasons['Date'].shape[0]):
    Week_day_numeric.append(reasons['Date'][i].weekday())
    Week_day_name.append(reasons['Date'][i].day_name())

In [237]:
reasons['Week_day_numeric'] = Week_day_numeric
reasons['Week_day'] = Week_day_name

In [238]:
reasons['Week_day_numeric'].head()

0    1
1    1
2    2
3    3
4    3
Name: Week_day_numeric, dtype: int64

In [239]:
reasons['Week_day'].head()

0      Tuesday
1      Tuesday
2    Wednesday
3     Thursday
4     Thursday
Name: Week_day, dtype: object

In [240]:
Year_list = []
for i in range(reasons['Date'].shape[0]):
    Year_list.append(reasons['Date'][i].year)


In [241]:
len(Year_list)

700

In [242]:
reasons['Year'] = Year_list

In [243]:
reasons.head()

Unnamed: 0,Reason_1,Reason_2,Reason_3,Reason_4,Date,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours,Month,Week_day_numeric,Week_day,Year
0,0,0,0,1,2015-07-07,289,36,33,239.554,30,1,2,1,4,July,1,Tuesday,2015
1,0,0,0,0,2015-07-14,118,13,50,239.554,31,1,1,0,0,July,1,Tuesday,2015
2,0,0,0,1,2015-07-15,179,51,38,239.554,31,1,0,0,2,July,2,Wednesday,2015
3,1,0,0,0,2015-07-16,279,5,39,239.554,24,1,2,0,4,July,3,Thursday,2015
4,0,0,0,1,2015-07-23,289,36,33,239.554,30,1,2,1,2,July,3,Thursday,2015


In [244]:
reasons.columns.values

array(['Reason_1', 'Reason_2', 'Reason_3', 'Reason_4', 'Date',
       'Transportation Expense', 'Distance to Work', 'Age',
       'Daily Work Load Average', 'Body Mass Index', 'Education',
       'Children', 'Pets', 'Absenteeism Time in Hours', 'Month',
       'Week_day_numeric', 'Week_day', 'Year'], dtype=object)

In [245]:
column_name =  ['Reason_1', 'Reason_2', 'Reason_3', 'Reason_4','Date', 'Year', 'Month', 'Week_day', 'Week_day_numeric',
       'Transportation Expense', 'Distance to Work', 'Age',
       'Daily Work Load Average', 'Body Mass Index', 'Education', 'Children',
       'Pets', 'Absenteeism Time in Hours']

In [246]:
reasons = reasons[column_name]

In [247]:
reasons.head()

Unnamed: 0,Reason_1,Reason_2,Reason_3,Reason_4,Date,Year,Month,Week_day,Week_day_numeric,Transportation Expense,Distance to Work,Age,Daily Work Load Average,Body Mass Index,Education,Children,Pets,Absenteeism Time in Hours
0,0,0,0,1,2015-07-07,2015,July,Tuesday,1,289,36,33,239.554,30,1,2,1,4
1,0,0,0,0,2015-07-14,2015,July,Tuesday,1,118,13,50,239.554,31,1,1,0,0
2,0,0,0,1,2015-07-15,2015,July,Wednesday,2,179,51,38,239.554,31,1,0,0,2
3,1,0,0,0,2015-07-16,2015,July,Thursday,3,279,5,39,239.554,24,1,2,0,4
4,0,0,0,1,2015-07-23,2015,July,Thursday,3,289,36,33,239.554,30,1,2,1,2


## Education

In [248]:
reasons['Education'].unique()

array([1, 3, 2, 4], dtype=int64)

In [249]:
reasons['Education'].value_counts()

1    583
3     73
2     40
4      4
Name: Education, dtype: int64

In [251]:
# categorizing the education column into two groups
reasons['Education'] = reasons['Education'].map({1:0, 2:1, 3:1, 4:1})

In [252]:
reasons['Education'].unique()

array([0, 1], dtype=int64)

In [253]:
reasons['Education'].value_counts()

0    583
1    117
Name: Education, dtype: int64

In [254]:
reasons['Year'].unique()

array([2015, 2016, 2017, 2018], dtype=int64)

## Saving the Dataset to export for future analysis

In [255]:
reasons.to_csv('reasons.csv')