## CH 02: Essential DataFrame Operations

In [81]:
import pandas as pd
import numpy as np
pd.set_option('display.max_columns', 10, 'display.max_rows', 10)

## Introduction

## Selecting Multiple DataFrame Columns

### How to do it\...

In [82]:
movies = pd.read_csv('../data/movie.csv')
movies.head()

Unnamed: 0,color,director_name,num_critic_for_reviews,duration,director_facebook_likes,...,title_year,actor_2_facebook_likes,imdb_score,aspect_ratio,movie_facebook_likes
0,Color,James Cameron,723.0,178.0,0.0,...,2009.0,936.0,7.9,1.78,33000
1,Color,Gore Verbinski,302.0,169.0,563.0,...,2007.0,5000.0,7.1,2.35,0
2,Color,Sam Mendes,602.0,148.0,0.0,...,2015.0,393.0,6.8,2.35,85000
3,Color,Christopher Nolan,813.0,164.0,22000.0,...,2012.0,23000.0,8.5,2.35,164000
4,,Doug Walker,,,131.0,...,,12.0,7.1,,0


In [83]:
# Read in the movie dataset, and pass in a list of the desired columns to the indexing operator

movie_actor_director = movies[['actor_1_name', 'actor_2_name', 'actor_3_name', 'director_name']]
movie_actor_director.head()

Unnamed: 0,actor_1_name,actor_2_name,actor_3_name,director_name
0,CCH Pounder,Joel David Moore,Wes Studi,James Cameron
1,Johnny Depp,Orlando Bloom,Jack Davenport,Gore Verbinski
2,Christoph Waltz,Rory Kinnear,Stephanie Sigman,Sam Mendes
3,Tom Hardy,Christian Bale,Joseph Gordon-Levitt,Christopher Nolan
4,Doug Walker,Rob Walker,,Doug Walker


### Selecting Cols as a series or as a data frame

In [84]:
# Here it is double [[]] brackets means a list. So we get a data frame here.

type(movies[['director_name']])

pandas.core.frame.DataFrame

In [85]:
# # Here it is double [] brackets means a list. So we get a data frame here.

type(movies['director_name'])

pandas.core.series.Series

In [86]:
# Using .loc method to get a data frame.

type(movies.loc[:, ['director_name']])

pandas.core.frame.DataFrame

In [7]:
# Using .loc method to get a series.

type(movies.loc[:, 'director_name'])

pandas.core.series.Series

### How it works\...

### There\'s more\...

In [10]:
# Passing a long list inside the indexing operator might cause readability issues. 
# To help with this, you may save all your column names to a list variable first.

cols = ['actor_1_name', 'actor_2_name',
        'actor_3_name', 'director_name']

movie_actor_director = movies[cols]

## Selecting Columns with Methods

### How it works\...

In [12]:
# Using methods like .select and .filter

movies.select_dtypes(include='int').head()

Unnamed: 0,num_voted_users,cast_total_facebook_likes,movie_facebook_likes
0,886204,4834,33000
1,471220,48350,0
2,275868,11700,85000
3,1144337,106759,164000
4,8,143,0


In [13]:
movies.select_dtypes(include='number').head()

Unnamed: 0,num_critic_for_reviews,duration,...,aspect_ratio,movie_facebook_likes
0,723.0,178.0,...,1.78,33000
1,302.0,169.0,...,2.35,0
2,602.0,148.0,...,2.35,85000
3,813.0,164.0,...,2.35,164000
4,,,...,,0


In [14]:
movies.select_dtypes(include=['int', 'object']).head()

Unnamed: 0,color,director_name,...,content_rating,movie_facebook_likes
0,Color,James Cameron,...,PG-13,33000
1,Color,Gore Verbinski,...,PG-13,0
2,Color,Sam Mendes,...,PG-13,85000
3,Color,Christopher Nolan,...,PG-13,164000
4,,Doug Walker,...,,0


In [15]:
movies.select_dtypes(exclude='float').head()

Unnamed: 0,color,director_name,...,content_rating,movie_facebook_likes
0,Color,James Cameron,...,PG-13,33000
1,Color,Gore Verbinski,...,PG-13,0
2,Color,Sam Mendes,...,PG-13,85000
3,Color,Christopher Nolan,...,PG-13,164000
4,,Doug Walker,...,,0


In [16]:
movies.filter(like='fb').head()

0
1
2
3
4


In [17]:
cols = ['actor_1_name', 'actor_2_name',
        'actor_3_name', 'director_name']
movies.filter(items=cols).head()

Unnamed: 0,actor_1_name,actor_2_name,actor_3_name,director_name
0,CCH Pounder,Joel David Moore,Wes Studi,James Cameron
1,Johnny Depp,Orlando Bloom,Jack Davenport,Gore Verbinski
2,Christoph Waltz,Rory Kinnear,Stephanie Sigman,Sam Mendes
3,Tom Hardy,Christian Bale,Joseph Gordon-Levitt,Christopher Nolan
4,Doug Walker,Rob Walker,,Doug Walker


In [18]:
movies.filter(regex=r'\d').head()

Unnamed: 0,actor_3_facebook_likes,actor_2_name,...,actor_3_name,actor_2_facebook_likes
0,855.0,Joel David Moore,...,Wes Studi,936.0
1,1000.0,Orlando Bloom,...,Jack Davenport,5000.0
2,161.0,Rory Kinnear,...,Stephanie Sigman,393.0
3,23000.0,Christian Bale,...,Joseph Gordon-Levitt,23000.0
4,,Rob Walker,...,,12.0


### How it works\...

### There\'s more\...

### See also

## Ordering Column Names

### How to do it\...

In [26]:
# Important

movies = pd.read_csv('../data/movie.csv')
def shorten(col):
    return (col.replace('facebook_likes', 'fb')
               .replace('_for_reviews', '')
    )
movies = movies.rename(columns=shorten)

In [27]:
movies.columns

Index(['color', 'director_name', 'num_critic', 'duration', 'director_fb',
       'actor_3_fb', 'actor_2_name', 'actor_1_fb', 'gross', 'genres',
       'actor_1_name', 'movie_title', 'num_voted_users', 'cast_total_fb',
       'actor_3_name', 'facenumber_in_poster', 'plot_keywords',
       'movie_imdb_link', 'num_user', 'language', 'country', 'content_rating',
       'budget', 'title_year', 'actor_2_fb', 'imdb_score', 'aspect_ratio',
       'movie_fb'],
      dtype='object')

In [28]:
cat_core = ['movie_title', 'title_year',
            'content_rating', 'genres']
cat_people = ['director_name', 'actor_1_name',
              'actor_2_name', 'actor_3_name']
cat_other = ['color', 'country', 'language',
             'plot_keywords', 'movie_imdb_link']
cont_fb = ['director_fb', 'actor_1_fb',
           'actor_2_fb', 'actor_3_fb',
           'cast_total_fb', 'movie_fb']
cont_finance = ['budget', 'gross']
cont_num_reviews = ['num_voted_users', 'num_user',
                    'num_critic']
cont_other = ['imdb_score', 'duration',
               'aspect_ratio', 'facenumber_in_poster']

In [29]:
new_col_order = cat_core + cat_people + \
                cat_other + cont_fb + \
                cont_finance + cont_num_reviews + \
                cont_other
set(movies.columns) == set(new_col_order)

True

In [30]:
movies[new_col_order].head()

Unnamed: 0,movie_title,title_year,...,aspect_ratio,facenumber_in_poster
0,Avatar,2009.0,...,1.78,0.0
1,Pirates of the Caribbean: At World's End,2007.0,...,2.35,0.0
2,Spectre,2015.0,...,2.35,1.0
3,The Dark Knight Rises,2012.0,...,2.35,0.0
4,Star Wars: Episode VII - The Force Awakens,,...,,0.0


### How it works\...

### There\'s more\...

### See also

## Summarizing a DataFrame

### How to do it\...

In [55]:
movies = pd.read_csv('../data/movie.csv')
movies.shape

(4916, 28)

In [32]:
movies.size

137648

In [33]:
movies.ndim

2

In [34]:
len(movies)

4916

In [35]:
movies.count()

color                      4897
director_name              4814
num_critic_for_reviews     4867
duration                   4901
director_facebook_likes    4814
                           ... 
title_year                 4810
actor_2_facebook_likes     4903
imdb_score                 4916
aspect_ratio               4590
movie_facebook_likes       4916
Length: 28, dtype: int64

In [37]:
movies.describe().T

Unnamed: 0,count,mean,...,75%,max
num_critic_for_reviews,4867.0,137.988905,...,191.00,813.0
duration,4901.0,107.090798,...,118.00,511.0
director_facebook_likes,4814.0,691.014541,...,189.75,23000.0
actor_3_facebook_likes,4893.0,631.276313,...,633.00,23000.0
actor_1_facebook_likes,4909.0,6494.488491,...,11000.00,640000.0
...,...,...,...,...,...
title_year,4810.0,2002.447609,...,2011.00,2016.0
actor_2_facebook_likes,4903.0,1621.923516,...,912.00,137000.0
imdb_score,4916.0,6.437429,...,7.20,9.5
aspect_ratio,4590.0,2.222349,...,2.35,16.0


In [38]:
movies.describe(percentiles=[.01, .3, .99]).T

Unnamed: 0,count,mean,...,99%,max
num_critic_for_reviews,4867.0,137.988905,...,546.68,813.0
duration,4901.0,107.090798,...,189.00,511.0
director_facebook_likes,4814.0,691.014541,...,16000.00,23000.0
actor_3_facebook_likes,4893.0,631.276313,...,11000.00,23000.0
actor_1_facebook_likes,4909.0,6494.488491,...,44920.00,640000.0
...,...,...,...,...,...
title_year,4810.0,2002.447609,...,2016.00,2016.0
actor_2_facebook_likes,4903.0,1621.923516,...,17000.00,137000.0
imdb_score,4916.0,6.437429,...,8.50,9.5
aspect_ratio,4590.0,2.222349,...,4.00,16.0


### How it works\...

### There\'s more\...

## Chaining DataFrame Methods

### How to do it\...

In [40]:
movies = pd.read_csv('../data/movie.csv')
def shorten(col):
    return (col.replace('facebook_likes', 'fb')
               .replace('_for_reviews', '')
    )
movies = movies.rename(columns=shorten)
movies.isnull().head()

Unnamed: 0,color,director_name,...,aspect_ratio,movie_fb
0,False,False,...,False,False
1,False,False,...,False,False
2,False,False,...,False,False
3,False,False,...,False,False
4,True,False,...,True,False


In [41]:
(movies
   .isnull()
   .sum()
   .head()
)

color             19
director_name    102
num_critic        49
duration          15
director_fb      102
dtype: int64

In [42]:
movies.isnull().sum().sum()

2656

In [43]:
movies.isnull().any().any()

True

### How it works\...

### There\'s more\...

In [46]:
with pd.option_context('max_colwidth', 20):
    movies.select_dtypes(['object']).fillna('').max()

In [47]:
with pd.option_context('max_colwidth', 20):
    (movies
        .select_dtypes(['object'])
        .fillna('')
        .max()
    )

### See also

## DataFrame Operations

In [56]:
colleges = pd.read_csv('../data/college.csv')

In [57]:
colleges.head()


Unnamed: 0,INSTNM,CITY,...,MD_EARN_WNE_P10,GRAD_DEBT_MDN_SUPP
0,Alabama A & M University,Normal,...,30300,33888.0
1,University of Alabama at Birmingham,Birmingham,...,39700,21941.5
2,Amridge University,Montgomery,...,40100,23370.0
3,University of Alabama in Huntsville,Huntsville,...,45500,24097.0
4,Alabama State University,Montgomery,...,26600,33118.5


In [53]:
# It will not work

colleges + 5

TypeError: can only concatenate str (not "int") to str

In [58]:
# To successfully use an operator with a DataFrame, first select homogeneous data. For this
# recipe, we will select all the columns that begin with 'UGDS_'. These columns represent the
# fraction of undergraduate students by race. 

# To get started, we import the data and use the institution name as the label for our index, and 
# then select the columns we desire with the .filter method

colleges = pd.read_csv('../data/college.csv', index_col='INSTNM')
college_ugds = colleges.filter(like='UGDS_')
college_ugds.head()

Unnamed: 0_level_0,UGDS_WHITE,UGDS_BLACK,...,UGDS_NRA,UGDS_UNKN
INSTNM,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1
Alabama A & M University,0.0333,0.9353,...,0.0059,0.0138
University of Alabama at Birmingham,0.5922,0.26,...,0.0179,0.01
Amridge University,0.299,0.4192,...,0.0,0.2715
University of Alabama in Huntsville,0.6988,0.1255,...,0.0332,0.035
Alabama State University,0.0158,0.9208,...,0.0243,0.0137


In [59]:
name = 'Northwest-Shoals Community College'
college_ugds.loc[name]

UGDS_WHITE    0.7912
UGDS_BLACK    0.1250
UGDS_HISP     0.0339
UGDS_ASIAN    0.0036
UGDS_AIAN     0.0088
UGDS_NHPI     0.0006
UGDS_2MOR     0.0012
UGDS_NRA      0.0033
UGDS_UNKN     0.0324
Name: Northwest-Shoals Community College, dtype: float64

In [60]:
college_ugds.loc[name].round(2)

UGDS_WHITE    0.79
UGDS_BLACK    0.12
UGDS_HISP     0.03
UGDS_ASIAN    0.00
UGDS_AIAN     0.01
UGDS_NHPI     0.00
UGDS_2MOR     0.00
UGDS_NRA      0.00
UGDS_UNKN     0.03
Name: Northwest-Shoals Community College, dtype: float64

In [61]:
(college_ugds.loc[name] + .0001).round(2)

UGDS_WHITE    0.79
UGDS_BLACK    0.13
UGDS_HISP     0.03
UGDS_ASIAN    0.00
UGDS_AIAN     0.01
UGDS_NHPI     0.00
UGDS_2MOR     0.00
UGDS_NRA      0.00
UGDS_UNKN     0.03
Name: Northwest-Shoals Community College, dtype: float64

In [62]:
college_ugds + .00501

Unnamed: 0_level_0,UGDS_WHITE,UGDS_BLACK,...,UGDS_NRA,UGDS_UNKN
INSTNM,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1
Alabama A & M University,0.03831,0.94031,...,0.01091,0.01881
University of Alabama at Birmingham,0.59721,0.26501,...,0.02291,0.01501
Amridge University,0.30401,0.42421,...,0.00501,0.27651
University of Alabama in Huntsville,0.70381,0.13051,...,0.03821,0.04001
Alabama State University,0.02081,0.92581,...,0.02931,0.01871
...,...,...,...,...,...
SAE Institute of Technology San Francisco,,,...,,
Rasmussen College - Overland Park,,,...,,
National Personal Training Institute of Cleveland,,,...,,
Bay Area Medical Academy - San Jose Satellite Location,,,...,,


In [63]:
(college_ugds + .00501) // .01

Unnamed: 0_level_0,UGDS_WHITE,UGDS_BLACK,...,UGDS_NRA,UGDS_UNKN
INSTNM,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1
Alabama A & M University,3.0,94.0,...,1.0,1.0
University of Alabama at Birmingham,59.0,26.0,...,2.0,1.0
Amridge University,30.0,42.0,...,0.0,27.0
University of Alabama in Huntsville,70.0,13.0,...,3.0,4.0
Alabama State University,2.0,92.0,...,2.0,1.0
...,...,...,...,...,...
SAE Institute of Technology San Francisco,,,...,,
Rasmussen College - Overland Park,,,...,,
National Personal Training Institute of Cleveland,,,...,,
Bay Area Medical Academy - San Jose Satellite Location,,,...,,


In [64]:
college_ugds_op_round = (college_ugds + .00501) // .01 / 100
college_ugds_op_round.head()

Unnamed: 0_level_0,UGDS_WHITE,UGDS_BLACK,...,UGDS_NRA,UGDS_UNKN
INSTNM,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1
Alabama A & M University,0.03,0.94,...,0.01,0.01
University of Alabama at Birmingham,0.59,0.26,...,0.02,0.01
Amridge University,0.3,0.42,...,0.0,0.27
University of Alabama in Huntsville,0.7,0.13,...,0.03,0.04
Alabama State University,0.02,0.92,...,0.02,0.01


In [65]:
college_ugds_round = (college_ugds + .00001).round(2)
college_ugds_round

Unnamed: 0_level_0,UGDS_WHITE,UGDS_BLACK,...,UGDS_NRA,UGDS_UNKN
INSTNM,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1
Alabama A & M University,0.03,0.94,...,0.01,0.01
University of Alabama at Birmingham,0.59,0.26,...,0.02,0.01
Amridge University,0.30,0.42,...,0.00,0.27
University of Alabama in Huntsville,0.70,0.13,...,0.03,0.04
Alabama State University,0.02,0.92,...,0.02,0.01
...,...,...,...,...,...
SAE Institute of Technology San Francisco,,,...,,
Rasmussen College - Overland Park,,,...,,
National Personal Training Institute of Cleveland,,,...,,
Bay Area Medical Academy - San Jose Satellite Location,,,...,,


In [66]:
college_ugds_op_round.equals(college_ugds_round)

True

### How it works\...

In [67]:
.045 + .005

0.049999999999999996

### There\'s more\...

In [68]:
college2 = (college_ugds
    .add(.00501) 
    .floordiv(.01) 
    .div(100)
)
college2.equals(college_ugds_op_round)

True

### See also

## Comparing Missing Values

In [None]:
np.nan == np.nan

In [None]:
None == None

In [None]:
np.nan > 5

In [None]:
5 > np.nan

In [None]:
np.nan != 5

### Getting ready

In [None]:
college = pd.read_csv('data/college.csv', index_col='INSTNM')
college_ugds = college.filter(like='UGDS_')

In [None]:
college_ugds == .0019

In [None]:
college_self_compare = college_ugds == college_ugds
college_self_compare.head()

In [None]:
college_self_compare.all()

In [None]:
(college_ugds == np.nan).sum()

In [None]:
college_ugds.isnull().sum()

In [None]:
college_ugds.equals(college_ugds)

### How it works\...

### There\'s more\...

In [None]:
college_ugds.eq(.0019)    # same as college_ugds == .0019

In [None]:
from pandas.testing import assert_frame_equal
assert_frame_equal(college_ugds, college_ugds) is None

## Transposing the direction of a DataFrame operation

### How to do it\...

In [None]:
college = pd.read_csv('data/college.csv', index_col='INSTNM')
college_ugds = college.filter(like='UGDS_')
college_ugds.head()

In [None]:
college_ugds.count()

In [None]:
college_ugds.count(axis='columns').head()

In [None]:
college_ugds.sum(axis='columns').head()

In [None]:
college_ugds.median(axis='index')

### How it works\...

### There\'s more\...

In [None]:
college_ugds_cumsum = college_ugds.cumsum(axis=1)
college_ugds_cumsum.head()

### See also

## Determining college campus diversity

In [69]:
pd.read_csv('../data/college_diversity.csv', index_col='School')

Unnamed: 0_level_0,Diversity Index
School,Unnamed: 1_level_1
"Rutgers University--Newark Newark, NJ",0.76
"Andrews University Berrien Springs, MI",0.74
"Stanford University Stanford, CA",0.74
"University of Houston Houston, TX",0.74
"University of Nevada--Las Vegas Las Vegas, NV",0.74
"University of San Francisco San Francisco, CA",0.74
"San Francisco State University San Francisco, CA",0.73
"University of Illinois--Chicago Chicago, IL",0.73
"New Jersey Institute of Technology Newark, NJ",0.72
"Texas Woman's University Denton, TX",0.72


### How to do it\...

In [70]:
college = pd.read_csv('../data/college.csv', index_col='INSTNM')
college_ugds = college.filter(like='UGDS_')

In [71]:
(college_ugds.isnull()
   .sum(axis='columns')
   .sort_values(ascending=False)
   .head()
)

INSTNM
Excel Learning Center-San Antonio South              9
Western State College of Law at Argosy University    9
Albany Law School                                    9
Albany Medical College                               9
A T Still University of Health Sciences              9
dtype: int64

In [72]:
college_ugds = college_ugds.dropna(how='all')
college_ugds.isnull().sum()

UGDS_WHITE    0
UGDS_BLACK    0
UGDS_HISP     0
UGDS_ASIAN    0
UGDS_AIAN     0
UGDS_NHPI     0
UGDS_2MOR     0
UGDS_NRA      0
UGDS_UNKN     0
dtype: int64

In [73]:
college_ugds.ge(.15)

Unnamed: 0_level_0,UGDS_WHITE,UGDS_BLACK,...,UGDS_NRA,UGDS_UNKN
INSTNM,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1
Alabama A & M University,False,True,...,False,False
University of Alabama at Birmingham,True,True,...,False,False
Amridge University,True,True,...,False,True
University of Alabama in Huntsville,True,False,...,False,False
Alabama State University,False,True,...,False,False
...,...,...,...,...,...
Hollywood Institute of Beauty Careers-West Palm Beach,True,True,...,False,False
Hollywood Institute of Beauty Careers-Casselberry,False,True,...,False,False
Coachella Valley Beauty College-Beaumont,True,False,...,False,False
Dewey University-Mayaguez,False,False,...,False,False


In [74]:
diversity_metric = college_ugds.ge(.15).sum(axis='columns')
diversity_metric.head()

INSTNM
Alabama A & M University               1
University of Alabama at Birmingham    2
Amridge University                     3
University of Alabama in Huntsville    1
Alabama State University               1
dtype: int64

In [75]:
diversity_metric.value_counts()

1    3042
2    2884
3     876
4      63
0       7
5       2
Name: count, dtype: int64

In [76]:
diversity_metric.sort_values(ascending=False).head()

INSTNM
Central Texas Beauty College-Temple                               5
Regency Beauty Institute-Austin                                   5
Westwood College-O'Hare Airport                                   4
Regency Beauty Institute-Pasadena                                 4
Soma Institute-The National School of Clinical Massage Therapy    4
dtype: int64

In [77]:
college_ugds.loc[['Regency Beauty Institute-Austin',
                   'Central Texas Beauty College-Temple']]

Unnamed: 0_level_0,UGDS_WHITE,UGDS_BLACK,...,UGDS_NRA,UGDS_UNKN
INSTNM,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1
Regency Beauty Institute-Austin,0.1867,0.2133,...,0.0,0.2667
Central Texas Beauty College-Temple,0.1616,0.2323,...,0.0,0.1515


In [78]:
us_news_top = ['Rutgers University-Newark',
                  'Andrews University',
                  'Stanford University',
                  'University of Houston',
                  'University of Nevada-Las Vegas']
diversity_metric.loc[us_news_top]

INSTNM
Rutgers University-Newark         4
Andrews University                3
Stanford University               3
University of Houston             3
University of Nevada-Las Vegas    3
dtype: int64

### How it works\...

### There\'s more\...

In [79]:
(college_ugds
   .max(axis=1)
   .sort_values(ascending=False)
   .head(10)
)

INSTNM
Caribbean University-Ponce                                        1.0
Brighton Institute of Cosmetology                                 1.0
Mesivta Torah Vodaath Rabbinical Seminary                         1.0
Rabbinical College Telshe                                         1.0
University of Puerto Rico-Mayaguez                                1.0
Haskell Indian Nations University                                 1.0
Lake Career and Technical Center                                  1.0
Leon Studio One School of Hair Design & Career Training Center    1.0
Dewey University-Hato Rey                                         1.0
Columbia Central University-Caguas                                1.0
dtype: float64

In [80]:
(college_ugds > .01).all(axis=1).any()

True

### The END