# Loading data into a Pandas Data Frame

In [20]:
import pandas as pd

# you can run shell commands in the jupyter notebook by starting with a !
!ls ../resources/week-1/datasets

#to convert a csv into a pandas dataframe call the pandas read_csv function
df = pd.read_csv('../resources/week-1/datasets/Admission_Predict.csv')
df.head()

Admission_Predict.csv  ferpa.txt	  winequality-red.csv
buddhist.txt	       nytimeshealth.txt


Unnamed: 0,Serial No.,GRE Score,TOEFL Score,University Rating,SOP,LOR,CGPA,Research,Chance of Admit
0,1,337,118,4,4.5,4.5,9.65,1,0.92
1,2,324,107,4,4.0,4.5,8.87,1,0.76
2,3,316,104,3,3.0,3.5,8.0,1,0.72
3,4,322,110,3,3.5,2.5,8.67,1,0.8
4,5,314,103,2,2.0,3.0,8.21,0,0.65


In [21]:
# by default pandas will add its own index column. 
# you can override this on loading by specifying index column from the dataset on load
df = pd.read_csv('../resources/week-1/datasets/Admission_Predict.csv',index_col = 0)
df.head()

Unnamed: 0_level_0,GRE Score,TOEFL Score,University Rating,SOP,LOR,CGPA,Research,Chance of Admit
Serial No.,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1,Unnamed: 6_level_1,Unnamed: 7_level_1,Unnamed: 8_level_1
1,337,118,4,4.5,4.5,9.65,1,0.92
2,324,107,4,4.0,4.5,8.87,1,0.76
3,316,104,3,3.0,3.5,8.0,1,0.72
4,322,110,3,3.5,2.5,8.67,1,0.8
5,314,103,2,2.0,3.0,8.21,0,0.65


In [36]:
# the columns can be renamed with a dictionary of old column name : new column name passed in to the rename function
new_df = df.rename (columns = { 'SOP':'Statement of Purpose',
                                'LOR':'Letter of Recommendation'})
new_df.head()

Unnamed: 0_level_0,GRE Score,TOEFL Score,University Rating,Statement of Purpose,LOR,CGPA,Research,Chance of Admit
Serial No.,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1,Unnamed: 6_level_1,Unnamed: 7_level_1,Unnamed: 8_level_1
1,337,118,4,4.5,4.5,9.65,1,0.92
2,324,107,4,4.0,4.5,8.87,1,0.76
3,316,104,3,3.0,3.5,8.0,1,0.72
4,322,110,3,3.5,2.5,8.67,1,0.8
5,314,103,2,2.0,3.0,8.21,0,0.65


In [38]:
# if a particular column name does not change (note LOR above) you can interrogate the column names
# with the dataframes 'columns' attribute
new_df.columns
#note that the LOR column name actually has an embedded space

Index(['GRE Score', 'TOEFL Score', 'University Rating', 'Statement of Purpose',
       'LOR ', 'CGPA', 'Research', 'Chance of Admit '],
      dtype='object')

In [50]:
# to handle this, you could pass in the extra space, but a more robust solution would
# be to make use of pythons 'strip' method that will remove all sorts of whitespace from
# the beginning or end of a string object
# this function is passed into the rename function using the 'mapper' parameter

new_df = df.rename( mapper=str.strip, axis='columns')
# If you try to combine these in one statement it tells you that you cannot specify
# both the mapper and the columns parameter at the same time
new_df = new_df.rename (columns = { 'SOP':'Statement of Purpose',
                                'LOR':'Letter of Recommendation'})
new_df.head()


Unnamed: 0_level_0,GRE Score,TOEFL Score,University Rating,Statement of Purpose,Letter of Recommendation,CGPA,Research,Chance of Admit
Serial No.,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1,Unnamed: 6_level_1,Unnamed: 7_level_1,Unnamed: 8_level_1
1,337,118,4,4.5,4.5,9.65,1,0.92
2,324,107,4,4.0,4.5,8.87,1,0.76
3,316,104,3,3.0,3.5,8.0,1,0.72
4,322,110,3,3.5,2.5,8.67,1,0.8
5,314,103,2,2.0,3.0,8.21,0,0.65


In [56]:
# badly formatted columns names is and extremely common problem with csv files
# a robust way to handle these is with list and list comprehension

# first make a list of your column names
cols = list(df.columns)
# run a comprenehnsion on the list
cols  = [x.lower().strip() for x in cols]
#replace the column names in the dataframe
df.columns = cols
df

Unnamed: 0_level_0,gre score,toefl score,university rating,sop,lor,cgpa,research,chance of admit
Serial No.,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1,Unnamed: 6_level_1,Unnamed: 7_level_1,Unnamed: 8_level_1
1,337,118,4,4.5,4.5,9.65,1,0.92
2,324,107,4,4.0,4.5,8.87,1,0.76
3,316,104,3,3.0,3.5,8.00,1,0.72
4,322,110,3,3.5,2.5,8.67,1,0.80
5,314,103,2,2.0,3.0,8.21,0,0.65
...,...,...,...,...,...,...,...,...
396,324,110,3,3.5,3.5,9.04,1,0.82
397,325,107,3,3.0,3.5,9.11,1,0.84
398,330,116,4,5.0,4.5,9.45,1,0.91
399,312,103,3,3.5,4.0,8.78,0,0.67
