In [4]:
import pandas as pd
import numpy as np

In [5]:
def reduce_mem_usage(df):
    start_mem_usg = df.memory_usage().sum() / 1024**2 
    print("Memory usage of properties dataframe is :",start_mem_usg," MB")
    NAlist = [] # Keeps track of columns that have missing values filled in. 
    for col in df.columns:
        if df[col].dtype != object:  # Exclude strings
            
            # Print current column type
            print("******************************")
            print("Column: ",col)
            print("dtype before: ",df[col].dtype)
            
            # make variables for Int, max and min
            IsInt = False
            mx = df[col].max()
            mn = df[col].min()
            
            # Integer does not support NA, therefore, NA needs to be filled
            if not np.isfinite(df[col]).all(): 
                continue
                NAlist.append(col)
                df[col].fillna(mn-1,inplace=True)  
                   
            # test if column can be converted to an integer
            asint = df[col].fillna(0).astype(np.int64)
            result = (df[col] - asint)
            result = result.sum()
            if result > -0.01 and result < 0.01:
                IsInt = True

            
            # Make Integer/unsigned Integer datatypes
            if IsInt:
                if mn >= 0:
                    if mx < 255:
                        df[col] = df[col].astype(np.uint8)
                    elif mx < 65535:
                        df[col] = df[col].astype(np.uint16)
                    elif mx < 4294967295:
                        df[col] = df[col].astype(np.uint32)
                    else:
                        df[col] = df[col].astype(np.uint64)
                else:
                    if mn > np.iinfo(np.int8).min and mx < np.iinfo(np.int8).max:
                        df[col] = df[col].astype(np.int8)
                    elif mn > np.iinfo(np.int16).min and mx < np.iinfo(np.int16).max:
                        df[col] = df[col].astype(np.int16)
                    elif mn > np.iinfo(np.int32).min and mx < np.iinfo(np.int32).max:
                        df[col] = df[col].astype(np.int32)
                    elif mn > np.iinfo(np.int64).min and mx < np.iinfo(np.int64).max:
                        df[col] = df[col].astype(np.int64)    
            
            # Make float datatypes 32 bit
            else:
                df[col] = df[col].astype(np.float32)
            
            # Print new column type
            print("dtype after: ",df[col].dtype)
            print("******************************")
    
    # Print final result
    print("___MEMORY USAGE AFTER COMPLETION:___")
    mem_usg = df.memory_usage().sum() / 1024**2 
    print("Memory usage is: ",mem_usg," MB")
    print("This is ",100*mem_usg/start_mem_usg,"% of the initial size")
    return df

In [6]:
df1 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_AL-IL_2012Q4.csv")
df1 = reduce_mem_usage(df1)
df2 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_IN-NY_2012Q4.csv")
df2 = reduce_mem_usage(df2)
df3 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_NC-WY_2012Q4.csv")
df3 = reduce_mem_usage(df3)

Memory usage of properties dataframe is : 2214.622772216797  MB
******************************
Column:  State Code
dtype before:  int64
dtype after:  uint8
******************************
******************************
Column:  County Code
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Model Year
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Min HP
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Max HP
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Min kW
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Max kW
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehi

AttributeError: module 'pandas' has no attribute 'concatenate'

In [17]:
df1 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_AL-IL_2014Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df1 = reduce_mem_usage(df1)
df2 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_IN-NY_2014Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df2 = reduce_mem_usage(df2)
df3 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_NC-WY_2014Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df3 = reduce_mem_usage(df3)

Memory usage of properties dataframe is : 2390.610939025879  MB
******************************
Column:  State Code
dtype before:  int64
dtype after:  uint8
******************************
******************************
Column:  County Code
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Model Year
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Wheelbase
dtype before:  float64
******************************
Column:  Vehicle Weight
dtype before:  float64
******************************
Column:  Vehicle Min HP
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Max HP
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Min kW
dtype before:  int64
dtype after:  uint16
******************************
*************

In [19]:
df1 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_AL-IL_2016Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df1 = reduce_mem_usage(df1)
df2 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_IN-NY_2016Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df2 = reduce_mem_usage(df2)
df3 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_NC-WY_2016Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df3 = reduce_mem_usage(df3)

Memory usage of properties dataframe is : 2508.8667907714844  MB
******************************
Column:  State Code
dtype before:  int64
dtype after:  uint8
******************************
******************************
Column:  County Code
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Model Year
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Wheelbase
dtype before:  float64
******************************
Column:  Vehicle Weight
dtype before:  float64
******************************
Column:  Vehicle Min HP
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Max HP
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Min kW
dtype before:  int64
dtype after:  uint16
******************************
************

In [None]:
df1 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_AL-IL_2018Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df1 = reduce_mem_usage(df1)
df2 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_IN-NY_2018Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df2 = reduce_mem_usage(df2)
df3 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_NC-WY_2018Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df3 = reduce_mem_usage(df3)

Memory usage of properties dataframe is : 2677.2658615112305  MB
******************************
Column:  State Code
dtype before:  int64
dtype after:  uint8
******************************
******************************
Column:  County Code
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Model Year
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Wheelbase
dtype before:  float64
******************************
Column:  Vehicle Weight
dtype before:  float64
******************************
Column:  Vehicle Min HP
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Max HP
dtype before:  int64
dtype after:  uint16
******************************
******************************
Column:  Vehicle Min kW
dtype before:  int64
dtype after:  uint16
******************************
************

In [None]:
df1 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_AL-IL_2020Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df1 = reduce_mem_usage(df1)
df2 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_IN-NY_2020Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df2 = reduce_mem_usage(df2)
df3 = pd.read_csv("../Data/Transport/Experian Registrations/US_EVD_Fleet_LD_COUNTY_NC-WY_2020Q4.csv",usecols=lambda x: x != 'Vehicle GVW Class')
df3 = reduce_mem_usage(df3)

In [None]:
df = pd.concat((df1,df2,df3),axis=0)
del df1
del df2
del df3
df.to_parquet("COUNTY_2018.parquet")
del df