### Import Modules

In [None]:
import pandas as pd
import pickle as pkl
import numpy as np

### Reduce Memory Usage

In [None]:
def reduce_mem_usage(props):
    start_mem_usg = props.memory_usage().sum() / 1024**2 
    print("Memory usage of properties dataframe is :",start_mem_usg," MB")
    NAlist = [] # Keeps track of columns that have missing values filled in. 
    for col in props.columns:
        if (props[col].dtype != object and props[col].dtype != 'datetime64[ns]'):  # Exclude strings
            
            # Print current column type
            print("******************************")
            print("Column: ",col)
            print("dtype before: ",props[col].dtype)
            
            # make variables for Int, max and min
            IsInt = False
            mx = props[col].max()
            mn = props[col].min()
            
            # Integer does not support NA, therefore, NA needs to be filled
            if not np.isfinite(props[col]).all(): 
                NAlist.append(col)
                props[col].fillna(mn-1,inplace=True)  
                   
            # test if column can be converted to an integer
            asint = props[col].fillna(0).astype(np.int64)
            result = (props[col] - asint)
            result = result.sum()
            if result > -0.01 and result < 0.01:
                IsInt = True

            
            # Make Integer/unsigned Integer datatypes
            if IsInt:
                if mn >= 0:
                    if mx < 255:
                        props[col] = props[col].astype(np.uint8)
                    elif mx < 65535:
                        props[col] = props[col].astype(np.uint16)
                    elif mx < 4294967295:
                        props[col] = props[col].astype(np.uint32)
                    else:
                        props[col] = props[col].astype(np.uint64)
                else:
                    if mn > np.iinfo(np.int8).min and mx < np.iinfo(np.int8).max:
                        props[col] = props[col].astype(np.int8)
                    elif mn > np.iinfo(np.int16).min and mx < np.iinfo(np.int16).max:
                        props[col] = props[col].astype(np.int16)
                    elif mn > np.iinfo(np.int32).min and mx < np.iinfo(np.int32).max:
                        props[col] = props[col].astype(np.int32)
                    elif mn > np.iinfo(np.int64).min and mx < np.iinfo(np.int64).max:
                        props[col] = props[col].astype(np.int64)    
            
            # Make float datatypes 32 bit
            else:
                props[col] = props[col].astype(np.float32)
            
            # Print new column type
            print("dtype after: ",props[col].dtype)
            print("******************************")
    
    # Print final result
    print("___MEMORY USAGE AFTER COMPLETION:___")
    mem_usg = props.memory_usage().sum() / 1024**2 
    print("Memory usage is: ",mem_usg," MB")
    print("This is ",100*mem_usg/start_mem_usg,"% of the initial size")
    return props

### Preapring Train Dataset

In [None]:
df_members = pkl.load(open('members.pkl','rb'))

In [None]:
df_train = pd.read_csv('/home/ubuntu/Project_3/kkbox-music-recommendation-challenge/train.csv', low_memory = True)

In [None]:
df_members = reduce_mem_usage(df_members)
df_train = reduce_mem_usage(df_train)

In [None]:
df_members_new = df_members.drop_duplicates(subset = 'msno', keep = 'first', inplace = False)
df_members_new.drop(columns = ['song_id', 'genre_ids'], inplace = True)

In [None]:
df_members_new.columns

In [None]:
df_train = df_train.merge(df_members_new, on = 'msno', how = 'left')

In [None]:
del df_members, df_members_new

In [None]:
df_train.columns

In [None]:
df_songs = pd.read_csv('/home/ubuntu/Project_3/kkbox-music-recommendation-challenge/songs.csv', low_memory = True)

In [None]:
df_train= df_train.merge(df_songs, on = 'song_id')

In [None]:
del df_songs

In [None]:
df_songs_extra = pd.read_csv('/home/ubuntu/Project_3/kkbox-music-recommendation-challenge/song_extra_info.csv', low_memory = True)

In [None]:
df_train = df_train.merge(df_songs_extra, on = 'song_id')

In [None]:
pkl.dump(df_train, open('train_complete', 'wb'))