In [2]:
import pandas as pd
import numpy as np
from pandas import DataFrame
from pandas import Series

# 处理缺失数据

## isnull()

In [2]:
string_data = Series(['aardvark', 'artichoke', np.nan, 'avocado'])
string_data

0     aardvark
1    artichoke
2          NaN
3      avocado
dtype: object

In [6]:
string_data.isnull()

0     True
1    False
2     True
3    False
dtype: bool

In [7]:
# None值会被当做NA处理

string_data[0] = None
string_data.isnull()

0     True
1    False
2     True
3    False
dtype: bool

## dropna()

### Series

In [9]:
data = Series([1, np.nan, 3.5, np.nan, 7])
data

0    1.0
1    NaN
2    3.5
3    NaN
4    7.0
dtype: float64

In [10]:
data.dropna()

0    1.0
2    3.5
4    7.0
dtype: float64

In [11]:
data[data.notnull()]

0    1.0
2    3.5
4    7.0
dtype: float64

### DataFrame
默认丢弃任何含有缺失值的行

#### 丢弃行

In [14]:
data = DataFrame([[1., 6.5, 3.], [1., np.nan, np.nan],
                     [np.nan, np.nan, np.nan], [np.nan, 6.5, 3.]])
data

Unnamed: 0,0,1,2
0,1.0,6.5,3.0
1,1.0,,
2,,,
3,,6.5,3.0


In [16]:
cleaned = data.dropna()
cleaned

Unnamed: 0,0,1,2
0,1.0,6.5,3.0


In [17]:
# how='all' 只丢弃全为NA的行
data.dropna(how='all')

Unnamed: 0,0,1,2
0,1.0,6.5,3.0
1,1.0,,
3,,6.5,3.0


#### 丢弃列

In [24]:
data[4] = np.nan
data

Unnamed: 0,0,1,2,4
0,1.0,6.5,3.0,
1,1.0,,,
2,,,,
3,,6.5,3.0,


In [25]:
data.dropna(axis='columns', how='all')

Unnamed: 0,0,1,2
0,1.0,6.5,3.0
1,1.0,,
2,,,
3,,6.5,3.0


In [44]:
df = DataFrame(np.random.randn(7, 3))
df.iloc[:4, 1] = np.nan
df.iloc[:2, 2] = np.nan
df

Unnamed: 0,0,1,2
0,1.023693,,
1,-2.122532,,
2,-1.155951,,-0.167326
3,0.778159,,-0.001092
4,0.373246,1.662954,1.871977
5,-0.31737,0.473683,-1.212071
6,-2.208364,0.800617,-0.25676


In [29]:
df.dropna()

Unnamed: 0,0,1,2
4,-0.107024,1.410166,0.766597
5,0.883597,0.05347,-1.423869
6,0.549759,-0.364777,0.5206


In [31]:
#  简单的理解：这一行除去NA值，剩余数值的数量大于等于n，便显示这一行。

df.dropna(thresh=2)

Unnamed: 0,0,1,2
2,-0.71201,,0.019109
3,0.92463,,-0.130632
4,-0.107024,1.410166,0.766597
5,0.883597,0.05347,-1.423869
6,0.549759,-0.364777,0.5206


## fillna()

In [32]:
df.fillna(0)

Unnamed: 0,0,1,2
0,0.102687,0.0,0.0
1,0.075442,0.0,0.0
2,-0.71201,0.0,0.019109
3,0.92463,0.0,-0.130632
4,-0.107024,1.410166,0.766597
5,0.883597,0.05347,-1.423869
6,0.549759,-0.364777,0.5206


In [36]:
# 通过字典调用fillna，对不同的列填充不同的值

df.fillna({1: 0.5, 2: 0})

Unnamed: 0,0,1,2
0,0.102687,0.5,0.0
1,0.075442,0.5,0.0
2,-0.71201,0.5,0.019109
3,0.92463,0.5,-0.130632
4,-0.107024,1.410166,0.766597
5,0.883597,0.05347,-1.423869
6,0.549759,-0.364777,0.5206


In [40]:
# 参数inplace=True对现有对象进行修改

df.fillna(0, inplace=True)
df

Unnamed: 0,0,1,2
0,0.102687,0.0,0.0
1,0.075442,0.0,0.0
2,-0.71201,0.0,0.019109
3,0.92463,0.0,-0.130632
4,-0.107024,1.410166,0.766597
5,0.883597,0.05347,-1.423869
6,0.549759,-0.364777,0.5206


In [48]:
df[1].fillna(df[1].mean())

0    0.979085
1    0.979085
2    0.979085
3    0.979085
4    1.662954
5    0.473683
6    0.800617
Name: 1, dtype: float64

In [52]:
df = DataFrame(np.random.randn(6, 3))
df.iloc[2:, 1] = np.nan
df.iloc[4:, 2] = np.nan
df

Unnamed: 0,0,1,2
0,0.411764,-1.29897,-0.737811
1,-0.593034,-0.818809,-0.002498
2,0.819622,,1.17547
3,1.235155,,-2.51211
4,-0.632379,,
5,1.15626,,


In [53]:
df.fillna(method='ffill')

Unnamed: 0,0,1,2
0,0.411764,-1.29897,-0.737811
1,-0.593034,-0.818809,-0.002498
2,0.819622,-0.818809,1.17547
3,1.235155,-0.818809,-2.51211
4,-0.632379,-0.818809,-2.51211
5,1.15626,-0.818809,-2.51211


In [55]:
# 参数limit，连续填充的最大数量

df.fillna(method='ffill', limit=2)

Unnamed: 0,0,1,2
0,0.411764,-1.29897,-0.737811
1,-0.593034,-0.818809,-0.002498
2,0.819622,-0.818809,1.17547
3,1.235155,-0.818809,-2.51211
4,-0.632379,,-2.51211
5,1.15626,,-2.51211


# 层次化索引

### series

In [3]:
data = Series(np.random.randn(10), 
              index=[['a' ,'a', 'a', 'b', 'b', 'b', 'c', 'c', 'd', 'd'],
                      [1, 2, 3, 1, 2, 3, 1, 2, 2, 3]]
             )
data

a  1    1.021059
   2    1.185696
   3   -0.389744
b  1   -0.303631
   2   -0.007666
   3   -0.704928
c  1   -0.173114
   2   -2.337327
d  2    0.044225
   3   -0.852355
dtype: float64

In [57]:
data.index

MultiIndex([('a', 1),
            ('a', 2),
            ('a', 3),
            ('b', 1),
            ('b', 2),
            ('b', 3),
            ('c', 1),
            ('c', 2),
            ('d', 2),
            ('d', 3)],
           )

In [59]:
data['b']

1    0.549786
2    0.124315
3    0.431137
dtype: float64

In [60]:
data['b':'c']

b  1    0.549786
   2    0.124315
   3    0.431137
c  1   -0.016035
   2   -0.630413
dtype: float64

In [62]:
data.loc[['b', 'd']]

b  1    0.549786
   2    0.124315
   3    0.431137
d  2   -2.446334
   3   -0.809636
dtype: float64

In [11]:
# 索引内层

data[:, 2]

a    1.185696
b   -0.007666
c   -2.337327
d    0.044225
dtype: float64

### DataFrame

In [15]:
frame = DataFrame(np.arange(12).reshape((4,3)), 
                  index=[['a', 'a', 'b', 'b'], [1,2,1,2]],
                  columns=[['Ohio' ,'Ohio', 'Colorado'],
                           ['Green', 'Red', 'Green']])
frame

Unnamed: 0_level_0,Unnamed: 1_level_0,Ohio,Ohio,Colorado
Unnamed: 0_level_1,Unnamed: 1_level_1,Green,Red,Green
a,1,0,1,2
a,2,3,4,5
b,1,6,7,8
b,2,9,10,11


In [17]:
frame.index.names = ['key1', 'key2']
frame.columns.names = ['state', 'color']
frame

Unnamed: 0_level_0,state,Ohio,Ohio,Colorado
Unnamed: 0_level_1,color,Green,Red,Green
key1,key2,Unnamed: 2_level_2,Unnamed: 3_level_2,Unnamed: 4_level_2
a,1,0,1,2
a,2,3,4,5
b,1,6,7,8
b,2,9,10,11


In [22]:
frame['Ohio', 'Green']

key1  key2
a     1       0
      2       3
b     1       6
      2       9
Name: (Ohio, Green), dtype: int32

## unstack()

In [14]:
# 在生成数据透视表的时候用

data.unstack()

Unnamed: 0,1,2,3
a,1.021059,1.185696,-0.389744
b,-0.303631,-0.007666,-0.704928
c,-0.173114,-2.337327,
d,,0.044225,-0.852355


## stack()

In [13]:
data.unstack().stack()

a  1    1.021059
   2    1.185696
   3   -0.389744
b  1   -0.303631
   2   -0.007666
   3   -0.704928
c  1   -0.173114
   2   -2.337327
d  2    0.044225
   3   -0.852355
dtype: float64

## 重排分级顺序
swaplevel

In [23]:
frame

Unnamed: 0_level_0,state,Ohio,Ohio,Colorado
Unnamed: 0_level_1,color,Green,Red,Green
key1,key2,Unnamed: 2_level_2,Unnamed: 3_level_2,Unnamed: 4_level_2
a,1,0,1,2
a,2,3,4,5
b,1,6,7,8
b,2,9,10,11


In [24]:
frame.swaplevel('key1', 'key2')

Unnamed: 0_level_0,state,Ohio,Ohio,Colorado
Unnamed: 0_level_1,color,Green,Red,Green
key2,key1,Unnamed: 2_level_2,Unnamed: 3_level_2,Unnamed: 4_level_2
1,a,0,1,2
2,a,3,4,5
1,b,6,7,8
2,b,9,10,11


In [28]:
frame.swaplevel(0, 1)

Unnamed: 0_level_0,state,Ohio,Ohio,Colorado
Unnamed: 0_level_1,color,Green,Red,Green
key2,key1,Unnamed: 2_level_2,Unnamed: 3_level_2,Unnamed: 4_level_2
1,a,0,1,2
2,a,3,4,5
1,b,6,7,8
2,b,9,10,11


In [27]:
frame.sort_index(1)

Unnamed: 0_level_0,state,Colorado,Ohio,Ohio
Unnamed: 0_level_1,color,Green,Green,Red
key1,key2,Unnamed: 2_level_2,Unnamed: 3_level_2,Unnamed: 4_level_2
a,1,2,0,1
a,2,5,3,4
b,1,8,6,7
b,2,11,9,10


## 根据级别汇总统计
参数level

In [29]:
frame.sum(level='key2')

state,Ohio,Ohio,Colorado
color,Green,Red,Green
key2,Unnamed: 1_level_2,Unnamed: 2_level_2,Unnamed: 3_level_2
1,6,8,10
2,12,14,16


In [30]:
frame.sum(level='color', axis=1)

Unnamed: 0_level_0,color,Green,Red
key1,key2,Unnamed: 2_level_1,Unnamed: 3_level_1
a,1,2,1
a,2,8,4
b,1,14,7
b,2,20,10


## 使用DataFrame的列
set_index

In [32]:
frame = DataFrame({'a':range(7),
                   'b':range(7,0,-1),
                   'c':['one', 'one', 'one', 'two', 'two', 'two', 'two'],
                   'd':[0,1,2,0,1,2,3]
    
})
frame

Unnamed: 0,a,b,c,d
0,0,7,one,0
1,1,6,one,1
2,2,5,one,2
3,3,4,two,0
4,4,3,two,1
5,5,2,two,2
6,6,1,two,3


In [35]:
frame2 = frame.set_index(['c', 'd'])
frame2

Unnamed: 0_level_0,Unnamed: 1_level_0,a,b
c,d,Unnamed: 2_level_1,Unnamed: 3_level_1
one,0,0,7
one,1,1,6
one,2,2,5
two,0,3,4
two,1,4,3
two,2,5,2
two,3,6,1


In [36]:
frame.set_index(['c', 'd'], drop=False)

Unnamed: 0_level_0,Unnamed: 1_level_0,a,b,c,d
c,d,Unnamed: 2_level_1,Unnamed: 3_level_1,Unnamed: 4_level_1,Unnamed: 5_level_1
one,0,0,7,one,0
one,1,1,6,one,1
one,2,2,5,one,2
two,0,3,4,two,0
two,1,4,3,two,1
two,2,5,2,two,2
two,3,6,1,two,3


In [37]:
# reset_index和set_index刚好相反
frame2.reset_index()

Unnamed: 0,c,d,a,b
0,one,0,0,7
1,one,1,1,6
2,one,2,2,5
3,two,0,3,4
4,two,1,4,3
5,two,2,5,2
6,two,3,6,1


# 其他

## 整数索引

### 整数索引

In [46]:
ser = Series(np.arange(3))
ser

0    0
1    1
2    2
dtype: int32

In [47]:
ser[-1]

KeyError: -1

### 非整数索引

In [42]:
ser2 = Series(np.arange(3), index=['a', 'b', 'c'])

ser2[-1]

2

In [53]:
# 如果轴索引含有索引器，数据选取的操作总是面向标签的
ser.loc[:1]

0    0
1    1
dtype: int32

## 面板数据