## How to Treat Missing Values in Data in Python

By Default,missing values are represented with NaN(Not a number)
!!!Warning : If your dataset has 0s,99s or 999s,be sure to either drop or approximate them as you would with missing values.

In [2]:
import numpy as np
import pandas as pd
from pandas import Series,DataFrame

### Figuring out what data is missing

In [4]:
missing = np.nan
series_obj = Series(['row1','row2',missing,'row4','row6',missing,'row8'])
series_obj

0    row1
1    row2
2     NaN
3    row4
4    row6
5     NaN
6    row8
dtype: object

In [5]:
series_obj.isnull()

0    False
1    False
2     True
3    False
4    False
5     True
6    False
dtype: bool

### Filling in for missing values

In [8]:
np.random.seed(25)
df_obj = DataFrame(np.random.rand(36).reshape(6,6))
df_obj

Unnamed: 0,0,1,2,3,4,5
0,0.870124,0.582277,0.278839,0.185911,0.4111,0.117376
1,0.684969,0.437611,0.556229,0.36708,0.402366,0.113041
2,0.447031,0.585445,0.161985,0.520719,0.326051,0.699186
3,0.366395,0.836375,0.481343,0.516502,0.383048,0.997541
4,0.514244,0.559053,0.03445,0.71993,0.421004,0.436935
5,0.281701,0.900274,0.669612,0.456069,0.289804,0.525819


In [11]:
df_obj.loc[3:5,0] = missing
df_obj.loc[1:4,5] = missing
df_obj

Unnamed: 0,0,1,2,3,4,5
0,0.870124,0.582277,0.278839,0.185911,0.4111,0.117376
1,0.684969,0.437611,0.556229,0.36708,0.402366,
2,0.447031,0.585445,0.161985,0.520719,0.326051,
3,,0.836375,0.481343,0.516502,0.383048,
4,,0.559053,0.03445,0.71993,0.421004,
5,,0.900274,0.669612,0.456069,0.289804,0.525819


#### Kayıp değerlerin yerine 0 değeri verme

In [12]:
filling_df = df_obj.fillna(0)
filling_df

Unnamed: 0,0,1,2,3,4,5
0,0.870124,0.582277,0.278839,0.185911,0.4111,0.117376
1,0.684969,0.437611,0.556229,0.36708,0.402366,0.0
2,0.447031,0.585445,0.161985,0.520719,0.326051,0.0
3,0.0,0.836375,0.481343,0.516502,0.383048,0.0
4,0.0,0.559053,0.03445,0.71993,0.421004,0.0
5,0.0,0.900274,0.669612,0.456069,0.289804,0.525819


#### Kayıp değerlerin yerine istenilen sayıları verme.

In [13]:
filled_df = df_obj.fillna({0:0.1,5:1.25})
filled_df

Unnamed: 0,0,1,2,3,4,5
0,0.870124,0.582277,0.278839,0.185911,0.4111,0.117376
1,0.684969,0.437611,0.556229,0.36708,0.402366,1.25
2,0.447031,0.585445,0.161985,0.520719,0.326051,1.25
3,0.1,0.836375,0.481343,0.516502,0.383048,1.25
4,0.1,0.559053,0.03445,0.71993,0.421004,1.25
5,0.1,0.900274,0.669612,0.456069,0.289804,0.525819


#### Kayıp değerlerin yerini en son bilinen değerlerle doldurma.

In [14]:
fill_df = df_obj.fillna(method = 'ffill')
fill_df

Unnamed: 0,0,1,2,3,4,5
0,0.870124,0.582277,0.278839,0.185911,0.4111,0.117376
1,0.684969,0.437611,0.556229,0.36708,0.402366,0.117376
2,0.447031,0.585445,0.161985,0.520719,0.326051,0.117376
3,0.447031,0.836375,0.481343,0.516502,0.383048,0.117376
4,0.447031,0.559053,0.03445,0.71993,0.421004,0.117376
5,0.447031,0.900274,0.669612,0.456069,0.289804,0.525819


### Counting Missing Values

In [15]:
np.random.seed(25)
df_obj = DataFrame(np.random.rand(36).reshape(6,6))
df_obj.loc[3:5,0] = missing
df_obj.loc[1:4,5] = missing
df_obj

Unnamed: 0,0,1,2,3,4,5
0,0.870124,0.582277,0.278839,0.185911,0.4111,0.117376
1,0.684969,0.437611,0.556229,0.36708,0.402366,
2,0.447031,0.585445,0.161985,0.520719,0.326051,
3,,0.836375,0.481343,0.516502,0.383048,
4,,0.559053,0.03445,0.71993,0.421004,
5,,0.900274,0.669612,0.456069,0.289804,0.525819


#### Her bir sütunda ne kadar kayıp değer olduğunu gösterir.

In [16]:
df_obj.isnull().sum()

0    3
1    0
2    0
3    0
4    0
5    4
dtype: int64

### Filtering out missing values

#### Kayıp değerlerin olduğu sütunu tümüyle siler.

In [17]:
df_no_nan = df_obj.dropna(axis=1)
df_no_nan

Unnamed: 0,1,2,3,4
0,0.582277,0.278839,0.185911,0.4111
1,0.437611,0.556229,0.36708,0.402366
2,0.585445,0.161985,0.520719,0.326051
3,0.836375,0.481343,0.516502,0.383048
4,0.559053,0.03445,0.71993,0.421004
5,0.900274,0.669612,0.456069,0.289804


In [18]:
df_obj.dropna(how='all')

Unnamed: 0,0,1,2,3,4,5
0,0.870124,0.582277,0.278839,0.185911,0.4111,0.117376
1,0.684969,0.437611,0.556229,0.36708,0.402366,
2,0.447031,0.585445,0.161985,0.520719,0.326051,
3,,0.836375,0.481343,0.516502,0.383048,
4,,0.559053,0.03445,0.71993,0.421004,
5,,0.900274,0.669612,0.456069,0.289804,0.525819
