In [1]:
import pandas as pd
import numpy as np

In [2]:
### Get all the pillar names from the excel

In [3]:
names = pd.read_excel('../../UNDP Digital Assessment Data Framework Filename Matching V7.xlsx')

In [4]:
col_names = ['Indicator','check', 'Data Source','Index','Filename']

In [5]:
names = names[col_names]

In [6]:
names.head()

Unnamed: 0,Indicator,check,Data Source,Index,Filename
0,Countries,,United Nations,False,Countries
1,"Database of Global Administrative Areas (GADM,...",,GADM maps and data,False,
2,High Resolution Population Density Maps + Demo...,,Facebook,False,
3,population density vs openstreetmap object den...,,Kontur,False,
4,Population Density,Infrastructure,World Bank,False,population_density


In [7]:
# get all the files per pillar
data_stats = names.groupby('check').agg({'Filename':'count','Indicator':'count'})

In [8]:
data_stats

Unnamed: 0_level_0,Filename,Indicator
check,Unnamed: 1_level_1,Unnamed: 2_level_1
Business,20,25
Foundations,9,12
Government,11,15
Infrastructure,45,48
People,38,46
Regulation,6,7
Strategy,1,1


In [9]:
### People

In [12]:
bnames = names[(names.check=='People')&(~names.Filename.isna())]#&(names.Index==False)]
bnames

Unnamed: 0,Indicator,check,Data Source,Index,Filename
99,Human Capital Index (HCI),People,DESA,True,e_government_index
100,% of population using internet (all),People,World Bank,False,population_using_internet
101,% of population using internet (female),People,ITU,False,gender_gap_internet_usage
102,% of population using internet (male),People,ITU,False,gender_gap_internet_usage
103,SDG 4.4 Digital literacy data,People,UNESCO,False,SDG 4.4_Digital_literacy_data
104,UNDP Human Development Index (HDI),People,UNDP,True,undp_human_developmnt
105,Facebook Social Connectedness Index,People,Facebook,True,fb_social_connectedness
106,Share of individuals using the Internet to int...,People,OECD,False,population_interacting_public_officials
107,Level of satisfaction for online public servic...,People,Boston Consulting Group/SalesForce,False,digital_public_service_use
108,Number of mobile apps available in national la...,People,GSMA Mobile Connectivity Index,False,apps_in_national_language


In [14]:
# get list of names for all indicators
indicators = bnames.Indicator.unique()

In [15]:
# get all file names
bfiles = bnames.Filename.unique()

In [16]:
bfiles

array(['e_government_index', 'population_using_internet',
       'gender_gap_internet_usage', 'SDG 4.4_Digital_literacy_data',
       'undp_human_developmnt', 'fb_social_connectedness',
       'population_interacting_public_officials',
       'digital_public_service_use', 'apps_in_national_language',
       'time_spent_online', ' happiness_score', 'cryptocurrency_adoption',
       'not_buying_online_concern_about_returning',
       'not_buying_online_concern_about_security',
       'ewaste_per_inhabitant', 'automation_led_unemployment',
       'cyberbullying_rate', 'global_wellbeing_initiative ',
       'financial_inclusiveness ',
       'individuals buying online and frequency', 'e-commerce_activity',
       'top_sites', 'youtube_searches', 'google_trends', 'intenet_usage',
       'household_internet_access', 'FB_users', 'gender_gaps',
       'population_digital_financial_services',
       'mobile_broadband_pricing', 'tax_percent_mobile_ownership',
       'percent_mobile_subscription'

In [17]:
# formula for converting scale
def convert_rank(old_value, old_min=1, old_max=7, new_min=1, new_max=6 ):
    """ Convert old scale values scale into new scale values"""
    old_range = old_max - old_min
    new_range = new_max - new_min
    new_value = (((old_value-old_min)*new_range)/old_range)+new_min
    return new_value

In [18]:
### 1. Human Capital Index (HCI)

In [21]:
indicators[0]

# load data
indicator = indicators[0]
print(indicator)
bf = bnames[bnames['Indicator']==indicator]['Filename'].values[0]
print(bf)

df = pd.read_csv('../../processed/{}.csv'.format(bf))

Human Capital Index (HCI)
e_government_index


In [23]:
df.head(10)

Unnamed: 0,Survey Year,Country Name,E-Government Rank,E-Government Index,E-Participation Index,Online Service Index,Human Capital Index,Telecommunication Infrastructure Index
0,2020,Iraq,143,0.436,0.3095,0.3353,0.4358,0.537
1,2020,Ireland,27,0.8433,0.8571,0.7706,0.9494,0.81
2,2020,Israel,30,0.8361,0.7143,0.7471,0.8924,0.8689
3,2020,Italy,37,0.8231,0.8214,0.8294,0.8466,0.7932
4,2020,Jamaica,114,0.5392,0.369,0.3882,0.7142,0.5151
5,2020,Japan,14,0.8989,0.9881,0.9059,0.8684,0.9223
6,2020,Jordan,117,0.5309,0.3333,0.3588,0.68,0.554
7,2020,Kazakhstan,29,0.8375,0.881,0.9235,0.8866,0.7024
8,2020,Kenya,116,0.5326,0.5952,0.6765,0.5812,0.3402
9,2020,Kiribati,145,0.432,0.5595,0.4941,0.6778,0.1241


In [25]:
# create standard columns
# df.rename(columns={'COUNTRY/ECONOMY':'Country Name'}, inplace=True)
df['higher_is_better'] = True
df['Indicator'] = indicator
df['data_col'] = df['Human Capital Index'] 
df['Year'] = df['Survey Year']

min_rank = df['data_col'].min()
max_rank = df['data_col'].max()

# transform 0-1 rank into 1-6
df['new_rank_score'] = df['data_col'].apply(lambda row: convert_rank(row,old_min=min_rank,old_max=max_rank))

In [27]:
df[['Country Name','Year','Indicator','data_col','new_rank_score','higher_is_better']].head(15)

Unnamed: 0,Country Name,Year,Indicator,data_col,new_rank_score,higher_is_better
0,Iraq,2020,Human Capital Index (HCI),0.4358,3.179,True
1,Ireland,2020,Human Capital Index (HCI),0.9494,5.747,True
2,Israel,2020,Human Capital Index (HCI),0.8924,5.462,True
3,Italy,2020,Human Capital Index (HCI),0.8466,5.233,True
4,Jamaica,2020,Human Capital Index (HCI),0.7142,4.571,True
5,Japan,2020,Human Capital Index (HCI),0.8684,5.342,True
6,Jordan,2020,Human Capital Index (HCI),0.68,4.4,True
7,Kazakhstan,2020,Human Capital Index (HCI),0.8866,5.433,True
8,Kenya,2020,Human Capital Index (HCI),0.5812,3.906,True
9,Kiribati,2020,Human Capital Index (HCI),0.6778,4.389,True


In [28]:
### 2. % of population using internet (all)

In [30]:
indicators[1]

# load data
indicator = indicators[1]
print(indicator)
bf = bnames[bnames['Indicator']==indicator]['Filename'].values[0]
print(bf)

df = pd.read_csv('../../processed/{}.csv'.format(bf))

% of population using internet (all)
population_using_internet


FileNotFoundError: [Errno 2] No such file or directory: '../../processed/population_using_internet.csv'

In [31]:
## 3. % of population using internet (female)

In [32]:
indicators[2]

# load data
indicator = indicators[2]
print(indicator)
bf = bnames[bnames['Indicator']==indicator]['Filename'].values[0]
print(bf)

df = pd.read_csv('../../processed/{}.csv'.format(bf))

% of population using internet (female)
gender_gap_internet_usage


In [33]:
df.head(10)
# Need to move the top row down one

Unnamed: 0.1,Unnamed: 0,Latest,All,Gender,Urban,Rural,data_country,data_year
0,,year,Individuals,Male,Total,Total,,
1,,2020,72.2,73.2,...,...,,
2,,2018,49.0,55.1,56.2,34.1,,
3,,2017,91.6,92.9,...,...,,
4,,2017,74.3,75.2,74.3,...,,
5,,2019,66.5,65.8,70.8,59.9,,
6,,2017,86.5,87.0,87.0,82.8,,
7,,2020,87.5,89.2,...,...,,
8,,2019,81.1,84.2,92.8,67.7,,
9,,2020,99.5,99.4,...,...,,
