# Handling Missing Data

In [14]:
import pandas as pd
df = pd.read_csv(r"C:\Users\amish\Desktop\weather_data.csv")
df
type(df.Date[0]) # elements under Date field being treated as strings


str

In [21]:
import pandas as pd
df = pd.read_csv(r"C:\Users\amish\Desktop\weather_data.csv", parse_dates=["Date"])
df
type(df.Date[0]) # elements under Dates now being treated as timestamps


pandas._libs.tslibs.timestamps.Timestamp

In [22]:
df.set_index("Date",inplace=True) #inplace=True edits the original data frame 
df

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,,9.0,Sunny
2017-01-05,28.0,,Snow
2017-01-06,,7.0,
2017-01-07,32.0,,Rain
2017-01-08,,,Sunny
2017-01-09,,,
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


# Using df.fillna() 

In [23]:
ef = df.fillna(0)
ef # fills not available values with passed value

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,0.0,9.0,Sunny
2017-01-05,28.0,0.0,Snow
2017-01-06,0.0,7.0,0
2017-01-07,32.0,0.0,Rain
2017-01-08,0.0,0.0,Sunny
2017-01-09,0.0,0.0,0
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


In [25]:
ef = df.fillna({
    "Temperature": 0,
    "Windspeed": 0,
    "Event": "No Event"
}) # passing a dictionary to specify the kind of data to be filled in place of missing values
ef

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,0.0,9.0,Sunny
2017-01-05,28.0,0.0,Snow
2017-01-06,0.0,7.0,No Event
2017-01-07,32.0,0.0,Rain
2017-01-08,0.0,0.0,Sunny
2017-01-09,0.0,0.0,No Event
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


In [26]:
ef = df.fillna(method="ffill")# forward fill fills the context of one cell into consecutive missing valued cells, similarly, bfill does the opposite
ef


Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,32.0,9.0,Sunny
2017-01-05,28.0,9.0,Snow
2017-01-06,28.0,7.0,Snow
2017-01-07,32.0,7.0,Rain
2017-01-08,32.0,7.0,Sunny
2017-01-09,32.0,7.0,Sunny
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


# Guessing appropriate values using interpolation

In [29]:
df

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,,9.0,Sunny
2017-01-05,28.0,,Snow
2017-01-06,,7.0,
2017-01-07,32.0,,Rain
2017-01-08,,,Sunny
2017-01-09,,,
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


In [30]:
ef = df.interpolate() #fills intermediate appropriate value for missing cells considering the values above and below
ef

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,30.0,9.0,Sunny
2017-01-05,28.0,8.0,Snow
2017-01-06,30.0,7.0,
2017-01-07,32.0,7.25,Rain
2017-01-08,32.666667,7.5,Sunny
2017-01-09,33.333333,7.75,
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


In [32]:
ef = df.interpolate(method="time")# sincle data for 2017-01-03/02 were not available so data for 5th was evaluated by even considering the ones on those dates
ef

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,29.0,9.0,Sunny
2017-01-05,28.0,8.0,Snow
2017-01-06,30.0,7.0,
2017-01-07,32.0,7.25,Rain
2017-01-08,32.666667,7.5,Sunny
2017-01-09,33.333333,7.75,
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


# Dropping rows with NaN values 

In [33]:
df

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,,9.0,Sunny
2017-01-05,28.0,,Snow
2017-01-06,,7.0,
2017-01-07,32.0,,Rain
2017-01-08,,,Sunny
2017-01-09,,,
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


In [34]:
ef = df.dropna() # drops all the rows with even atleast one NaN
ef

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


In [36]:
ef = df.dropna(how="all")# removes the rows with all the fields as NaN; not including index
ef

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,,9.0,Sunny
2017-01-05,28.0,,Snow
2017-01-06,,7.0,
2017-01-07,32.0,,Rain
2017-01-08,,,Sunny
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


In [39]:
ef = df.dropna(thresh=1)#thresh leaves all the rows with alteast one valid value; eliminating the rest
ef

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,,9.0,Sunny
2017-01-05,28.0,,Snow
2017-01-06,,7.0,
2017-01-07,32.0,,Rain
2017-01-08,,,Sunny
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


In [40]:
ef = df.dropna(thresh=2)#thresh leaves all the rows with alteast two valid value; eliminating the rest
ef

Unnamed: 0_level_0,Temperature,Windspeed,Event
Date,Unnamed: 1_level_1,Unnamed: 2_level_1,Unnamed: 3_level_1
2017-01-01,32.0,6.0,Rain
2017-01-04,,9.0,Sunny
2017-01-05,28.0,,Snow
2017-01-07,32.0,,Rain
2017-01-10,34.0,8.0,Cloudy
2017-01-11,40.0,12.0,Sunny


# Adding row(s) of data

In [68]:
dt = pd.date_range("2017-01-01","2017-01-11")
idx = pd.DatetimeIndex(dt)
df = df.reindex(idx)
df

Unnamed: 0,Temperature,Windspeed,Event
2017-01-01,32.0,6.0,Rain
2017-01-02,,,
2017-01-03,,,
2017-01-04,,9.0,Sunny
2017-01-05,28.0,,Snow
2017-01-06,,7.0,
2017-01-07,32.0,,Rain
2017-01-08,,,Sunny
2017-01-09,,,
2017-01-10,34.0,8.0,Cloudy
