# Import Relevant Libraries (Pandas and Numpy)

In [1]:
import pandas as pd
import numpy as np

## Read CSV File and display table using .head()

In [2]:
train_csv = pd.read_csv('train.csv')
train_csv.head()

Unnamed: 0,Id,MSSubClass,MSZoning,LotFrontage,LotArea,Street,Alley,LotShape,LandContour,Utilities,...,PoolArea,PoolQC,Fence,MiscFeature,MiscVal,MoSold,YrSold,SaleType,SaleCondition,SalePrice
0,1,60,RL,65.0,8450,Pave,,Reg,Lvl,AllPub,...,0,,,,0,2,2008,WD,Normal,208500
1,2,20,RL,80.0,9600,Pave,,Reg,Lvl,AllPub,...,0,,,,0,5,2007,WD,Normal,181500
2,3,60,RL,68.0,11250,Pave,,IR1,Lvl,AllPub,...,0,,,,0,9,2008,WD,Normal,223500
3,4,70,RL,60.0,9550,Pave,,IR1,Lvl,AllPub,...,0,,,,0,2,2006,WD,Abnorml,140000
4,5,60,RL,84.0,14260,Pave,,IR1,Lvl,AllPub,...,0,,,,0,12,2008,WD,Normal,250000


## Check the "shape" of the data

In [3]:
print("Data Dimensions: ", train_csv.shape)

Data Dimensions:  (1460, 81)


## Check the data types

In [4]:
print("Data types: ", train_csv.dtypes)

Data types:  Id                 int64
MSSubClass         int64
MSZoning          object
LotFrontage      float64
LotArea            int64
                  ...   
MoSold             int64
YrSold             int64
SaleType          object
SaleCondition     object
SalePrice          int64
Length: 81, dtype: object


## Use .info() method to see info on the data set (column names, data types)

In [5]:
train_csv.info()

<class 'pandas.core.frame.DataFrame'>
RangeIndex: 1460 entries, 0 to 1459
Data columns (total 81 columns):
 #   Column         Non-Null Count  Dtype  
---  ------         --------------  -----  
 0   Id             1460 non-null   int64  
 1   MSSubClass     1460 non-null   int64  
 2   MSZoning       1460 non-null   object 
 3   LotFrontage    1201 non-null   float64
 4   LotArea        1460 non-null   int64  
 5   Street         1460 non-null   object 
 6   Alley          91 non-null     object 
 7   LotShape       1460 non-null   object 
 8   LandContour    1460 non-null   object 
 9   Utilities      1460 non-null   object 
 10  LotConfig      1460 non-null   object 
 11  LandSlope      1460 non-null   object 
 12  Neighborhood   1460 non-null   object 
 13  Condition1     1460 non-null   object 
 14  Condition2     1460 non-null   object 
 15  BldgType       1460 non-null   object 
 16  HouseStyle     1460 non-null   object 
 17  OverallQual    1460 non-null   int64  
 18  OverallC

## Use the .describe() method to see different statistics of the data set

In [6]:
train_csv.describe()

Unnamed: 0,Id,MSSubClass,LotFrontage,LotArea,OverallQual,OverallCond,YearBuilt,YearRemodAdd,MasVnrArea,BsmtFinSF1,...,WoodDeckSF,OpenPorchSF,EnclosedPorch,3SsnPorch,ScreenPorch,PoolArea,MiscVal,MoSold,YrSold,SalePrice
count,1460.0,1460.0,1201.0,1460.0,1460.0,1460.0,1460.0,1460.0,1452.0,1460.0,...,1460.0,1460.0,1460.0,1460.0,1460.0,1460.0,1460.0,1460.0,1460.0,1460.0
mean,730.5,56.89726,70.049958,10516.828082,6.099315,5.575342,1971.267808,1984.865753,103.685262,443.639726,...,94.244521,46.660274,21.95411,3.409589,15.060959,2.758904,43.489041,6.321918,2007.815753,180921.19589
std,421.610009,42.300571,24.284752,9981.264932,1.382997,1.112799,30.202904,20.645407,181.066207,456.098091,...,125.338794,66.256028,61.119149,29.317331,55.757415,40.177307,496.123024,2.703626,1.328095,79442.502883
min,1.0,20.0,21.0,1300.0,1.0,1.0,1872.0,1950.0,0.0,0.0,...,0.0,0.0,0.0,0.0,0.0,0.0,0.0,1.0,2006.0,34900.0
25%,365.75,20.0,59.0,7553.5,5.0,5.0,1954.0,1967.0,0.0,0.0,...,0.0,0.0,0.0,0.0,0.0,0.0,0.0,5.0,2007.0,129975.0
50%,730.5,50.0,69.0,9478.5,6.0,5.0,1973.0,1994.0,0.0,383.5,...,0.0,25.0,0.0,0.0,0.0,0.0,0.0,6.0,2008.0,163000.0
75%,1095.25,70.0,80.0,11601.5,7.0,6.0,2000.0,2004.0,166.0,712.25,...,168.0,68.0,0.0,0.0,0.0,0.0,0.0,8.0,2009.0,214000.0
max,1460.0,190.0,313.0,215245.0,10.0,9.0,2010.0,2010.0,1600.0,5644.0,...,857.0,547.0,552.0,508.0,480.0,738.0,15500.0,12.0,2010.0,755000.0


## Read data from the given website

In [7]:
html_csv = pd.read_html('https://en.wikipedia.org/wiki/2016_Summer_Olympics_medal_table')

In [8]:
print("HTML Tables: ", len(html_csv))

HTML Tables:  7


In [9]:
html_csv[2].head()

Unnamed: 0,Rank,NOC,Gold,Silver,Bronze,Total
0,1,United States,46,37,38,121
1,2,Great Britain,27,23,17,67
2,3,China,26,18,26,70
3,4,Russia,19,17,20,56
4,5,Germany,17,10,15,42


In [10]:
summer2016 = html_csv[2]

In [11]:
top20 = summer2016[0:20]
top20.head(20)

Unnamed: 0,Rank,NOC,Gold,Silver,Bronze,Total
0,1,United States,46,37,38,121
1,2,Great Britain,27,23,17,67
2,3,China,26,18,26,70
3,4,Russia,19,17,20,56
4,5,Germany,17,10,15,42
5,6,Japan,12,8,21,41
6,7,France,10,18,14,42
7,8,South Korea,9,3,9,21
8,9,Italy,8,12,8,28
9,10,Australia,8,11,10,29


In [12]:
adult_data = pd.read_csv('adult.data', header = None)
adult_data.head()

Unnamed: 0,0,1,2,3,4,5,6,7,8,9,10,11,12,13,14
0,39,State-gov,77516,Bachelors,13,Never-married,Adm-clerical,Not-in-family,White,Male,2174,0,40,United-States,<=50K
1,50,Self-emp-not-inc,83311,Bachelors,13,Married-civ-spouse,Exec-managerial,Husband,White,Male,0,0,13,United-States,<=50K
2,38,Private,215646,HS-grad,9,Divorced,Handlers-cleaners,Not-in-family,White,Male,0,0,40,United-States,<=50K
3,53,Private,234721,11th,7,Married-civ-spouse,Handlers-cleaners,Husband,Black,Male,0,0,40,United-States,<=50K
4,28,Private,338409,Bachelors,13,Married-civ-spouse,Prof-specialty,Wife,Black,Female,0,0,40,Cuba,<=50K


In [13]:
print("Data Shape: ", adult_data.shape)

Data Shape:  (32561, 15)


## Use .info() method to see the column names and data types in the table

In [14]:
adult_data.info()

<class 'pandas.core.frame.DataFrame'>
RangeIndex: 32561 entries, 0 to 32560
Data columns (total 15 columns):
 #   Column  Non-Null Count  Dtype 
---  ------  --------------  ----- 
 0   0       32561 non-null  int64 
 1   1       32561 non-null  object
 2   2       32561 non-null  int64 
 3   3       32561 non-null  object
 4   4       32561 non-null  int64 
 5   5       32561 non-null  object
 6   6       32561 non-null  object
 7   7       32561 non-null  object
 8   8       32561 non-null  object
 9   9       32561 non-null  object
 10  10      32561 non-null  int64 
 11  11      32561 non-null  int64 
 12  12      32561 non-null  int64 
 13  13      32561 non-null  object
 14  14      32561 non-null  object
dtypes: int64(6), object(9)
memory usage: 3.7+ MB


## Use .describe() method to see different statistics of the data

In [15]:
adult_data.describe()

Unnamed: 0,0,2,4,10,11,12
count,32561.0,32561.0,32561.0,32561.0,32561.0,32561.0
mean,38.581647,189778.4,10.080679,1077.648844,87.30383,40.437456
std,13.640433,105550.0,2.57272,7385.292085,402.960219,12.347429
min,17.0,12285.0,1.0,0.0,0.0,1.0
25%,28.0,117827.0,9.0,0.0,0.0,40.0
50%,37.0,178356.0,10.0,0.0,0.0,40.0
75%,48.0,237051.0,12.0,0.0,0.0,45.0
max,90.0,1484705.0,16.0,99999.0,4356.0,99.0


In [16]:
df_list = []

In [17]:
for i in range(2000, 2017, 4):
    df_list.append(pd.read_html('https://en.wikipedia.org/wiki/{}_Summer_Olympics_medal_table'.format(i)))

In [18]:
df_list[2][2].head()

Unnamed: 0,Rank,NOC,Gold,Silver,Bronze,Total
0,1,China*,48,22,30,100
1,2,United States,36,39,37,112
2,3,Russia,24,13,23,60
3,4,Great Britain,19,13,19,51
4,5,Germany,16,11,14,41


In [19]:
top20 = []

In [20]:
for df in df_list:
    top20.append(df[2][0:20])

In [21]:
top20[1].head(20)

Unnamed: 0,Rank,Nation,Gold,Silver,Bronze,Total
0,1,United States,36,39,26,101
1,2,China,32,17,14,63
2,3,Russia,28,26,36,90
3,4,Australia,17,16,17,50
4,5,Japan,16,9,12,37
5,6,Germany,13,16,20,49
6,7,France,11,9,13,33
7,8,Italy,10,11,11,32
8,9,South Korea,9,12,9,30
9,10,Great Britain,9,9,12,30
