# JSON examples and exercise
****
+ get familiar with packages for dealing with JSON
+ study examples with JSON strings and files 
+ work on exercise to be completed and submitted 
****
+ reference: http://pandas.pydata.org/pandas-docs/stable/io.html#io-json-reader
+ data source: http://jsonstudio.com/resources/
****

In [3]:
import pandas as pd

## imports for Python, Pandas

In [6]:
import json
from pandas.io.json import json_normalize

## JSON example, with string

+ demonstrates creation of normalized dataframes (tables) from nested json string
+ source: http://pandas.pydata.org/pandas-docs/stable/io.html#normalization

In [4]:
# define json string
data = [{'state': 'Florida', 
         'shortname': 'FL',
         'info': {'governor': 'Rick Scott'},
         'counties': [{'name': 'Dade', 'population': 12345},
                      {'name': 'Broward', 'population': 40000},
                      {'name': 'Palm Beach', 'population': 60000}]},
        {'state': 'Ohio',
         'shortname': 'OH',
         'info': {'governor': 'John Kasich'},
         'counties': [{'name': 'Summit', 'population': 1234},
                      {'name': 'Cuyahoga', 'population': 1337}]}]

In [7]:
# use normalization to create tables from nested element
json_normalize(data, 'counties')

Unnamed: 0,name,population
0,Dade,12345
1,Broward,40000
2,Palm Beach,60000
3,Summit,1234
4,Cuyahoga,1337


In [8]:
# further populate tables created from nested element
json_normalize(data, 'counties', ['state', 'shortname', ['info', 'governor']])

Unnamed: 0,name,population,info.governor,state,shortname
0,Dade,12345,Rick Scott,Florida,FL
1,Broward,40000,Rick Scott,Florida,FL
2,Palm Beach,60000,Rick Scott,Florida,FL
3,Summit,1234,John Kasich,Ohio,OH
4,Cuyahoga,1337,John Kasich,Ohio,OH


****
## JSON example, with file

+ demonstrates reading in a json file as a string and as a table
+ uses small sample file containing data about projects funded by the World Bank 
+ data source: http://jsonstudio.com/resources/

In [2]:
# load json as string
#json.load((open('data/world_bank_projects_less.json')))

NameError: name 'json' is not defined

In [3]:
# load as Pandas dataframe
sample_json_df = pd.read_json('data/world_bank_projects_less.json')
#sample_json_df

NameError: name 'pd' is not defined

****
## exercise

Using data in file 'data/world_bank_projects.json' and the techniques demonstrated above,
1. Find the 10 countries with most projects
2. Find the top 10 major project themes (using column 'mjtheme_namecode')
3. In 2. above you will notice that some entries have only the code and the name is missing. Create a dataframe with the missing names filled in.

In [8]:
import pandas as pd
import json
from pandas.io.json import json_normalize

In [9]:
json.load((open('data/world_bank_projects.json')));

In [10]:
df=pd.read_json('data/world_bank_projects.json')

In [11]:
df_projects=df.groupby('countryname').count().sort_values('_id', ascending=False)

In [12]:
# 10 countries with most projects

print(df_projects[:10]['_id'])

countryname
People's Republic of China         19
Republic of Indonesia              19
Socialist Republic of Vietnam      17
Republic of India                  16
Republic of Yemen                  13
People's Republic of Bangladesh    12
Nepal                              12
Kingdom of Morocco                 12
Republic of Mozambique             11
Africa                             11
Name: _id, dtype: int64


In [34]:
# questions 2
#df['mjtheme_namecode_str'] = df['mjtheme_namecode'].astype(str)
#print(df.groupby(['mjtheme_namecode_str']).count().sort_values('_id', ascending=False)[:10]['_id'])


In [38]:
# question 2: Find the top 10 major project themes (using column 'mjtheme_namecode')

def sort_theme(projects):
    codes=list()
    
    for row in projects:
        #codes_row=list()
        for index in range(len(row)):
            item = row[index]
            #if item['code'] not in codes_row:
                #codes_row.append(item['code'])
            codes.append(item['code'])
    
    codedic=list()


    for row in projects:

        for index in range(len(row)):
            item = row[index]
            if (item not in codedic) and (item['name'] is not ''):
                codedic.append(item)
            
    newdict=dict()
    dict1=dict()
    
    for i in range(len(codedic)):
        newdict=dict()
        newdict[codedic[i]['code']]=codedic[i]['name']
        dict1.update(newdict)
    
    themes=dict()
    
    for j in range(1,12): 
        themes[dict1[str(j)]]=codes.count(str(j))
        
    import operator
    themes_sorted=sorted(themes.items(), key=operator.itemgetter(1),reverse=True)
    
    return themes_sorted
   
    
sort_theme(df['mjtheme_namecode'])

[('Environment and natural resources management', 250),
 ('Rural development', 216),
 ('Human development', 210),
 ('Public sector governance', 199),
 ('Social protection and risk management', 168),
 ('Financial and private sector development', 146),
 ('Social dev/gender/inclusion', 130),
 ('Trade and integration', 77),
 ('Urban development', 50),
 ('Economic management', 38),
 ('Rule of law', 15)]

In [39]:
# question 3:In 2. above you will notice that some entries have only the code and the name is missing. Create a dataframe with the missing names filled in.

codedic=list()


for row in df['mjtheme_namecode']:

    for index in range(len(row)):
        item = row[index]
        
        if (item not in codedic) and (item['name'] is not ''):
            codedic.append(item)
            

print(codedic)


[{'code': '8', 'name': 'Human development'}, {'code': '1', 'name': 'Economic management'}, {'code': '6', 'name': 'Social protection and risk management'}, {'code': '5', 'name': 'Trade and integration'}, {'code': '2', 'name': 'Public sector governance'}, {'code': '11', 'name': 'Environment and natural resources management'}, {'code': '7', 'name': 'Social dev/gender/inclusion'}, {'code': '4', 'name': 'Financial and private sector development'}, {'code': '10', 'name': 'Rural development'}, {'code': '9', 'name': 'Urban development'}, {'code': '3', 'name': 'Rule of law'}]


In [40]:
newdict=dict()
dict1=dict()
for i in range(11):
    newdict=dict()
    newdict[codedic[i]['code']]=codedic[i]['name']
    dict1.update(newdict)
    

print(dict1)



{'8': 'Human development', '1': 'Economic management', '6': 'Social protection and risk management', '5': 'Trade and integration', '2': 'Public sector governance', '11': 'Environment and natural resources management', '7': 'Social dev/gender/inclusion', '4': 'Financial and private sector development', '10': 'Rural development', '9': 'Urban development', '3': 'Rule of law'}


In [41]:


for row in df['mjtheme_namecode']:
    for index in range(len(row)):
        item = row[index]
        if item['name']=='':
            item['name']=dict1.get(item['code'])
            #print(item)
print(df['mjtheme_namecode'])
            
        

0      [{'code': '8', 'name': 'Human development'}, {...
1      [{'code': '1', 'name': 'Economic management'},...
2      [{'code': '5', 'name': 'Trade and integration'...
3      [{'code': '7', 'name': 'Social dev/gender/incl...
4      [{'code': '5', 'name': 'Trade and integration'...
5      [{'code': '6', 'name': 'Social protection and ...
6      [{'code': '2', 'name': 'Public sector governan...
7      [{'code': '11', 'name': 'Environment and natur...
8      [{'code': '10', 'name': 'Rural development'}, ...
9      [{'code': '2', 'name': 'Public sector governan...
10     [{'code': '10', 'name': 'Rural development'}, ...
11     [{'code': '10', 'name': 'Rural development'}, ...
12     [{'code': '4', 'name': 'Financial and private ...
13     [{'code': '5', 'name': 'Trade and integration'...
14     [{'code': '6', 'name': 'Social protection and ...
15     [{'code': '10', 'name': 'Rural development'}, ...
16     [{'code': '10', 'name': 'Rural development'}, ...
17     [{'code': '8', 'name': '