In [3]:
from __future__ import absolute_import, division, print_function, unicode_literals

import numpy as np
import pandas as pd

import tensorflow as tf
tf.enable_eager_execution()

from tensorflow import feature_column
from tensorflow.keras import layers
from sklearn.model_selection import train_test_split

In [4]:
# Use Pandas to create a dataframe
URL = 'https://storage.googleapis.com/applied-dl/heart.csv'
dataframe = pd.read_csv(URL)
dataframe.head()

Unnamed: 0,age,sex,cp,trestbps,chol,fbs,restecg,thalach,exang,oldpeak,slope,ca,thal,target
0,63,1,1,145,233,1,2,150,0,2.3,3,0,fixed,0
1,67,1,4,160,286,0,2,108,1,1.5,2,3,normal,1
2,67,1,4,120,229,0,2,129,1,2.6,2,2,reversible,0
3,37,1,3,130,250,0,0,187,0,3.5,3,0,normal,0
4,41,0,2,130,204,0,2,172,0,1.4,1,0,normal,0


In [5]:
# Split the data into training, validation and test

train,test = train_test_split(dataframe, test_size=0.2)
train,val = train_test_split(train, test_size=0.2)

print(len(train), 'train examples')
print(len(val), 'validation examples')
print(len(test), 'test examples')

193 train examples
49 validation examples
61 test examples


In [7]:
# Create an input pipelin using tf.data

# A utility method to create a tf.data dataset from a 
# Pandas Dataframe
def df_to_dataset(dataframe, shuffle=True, batch_size=32):
    dataframe = dataframe.copy()
    labels = dataframe.pop('target')
    ds = tf.data.Dataset.from_tensor_slices((dict(dataframe), labels))
    if shuffle:
        ds = ds.shuffle(buffer_size=len(dataframe))
    ds = ds.batch(batch_size)
    return ds

In [8]:
batch_size=5 # A small batch sized is used for 
# demonstration purpose
train_ds = df_to_dataset(train, batch_size=batch_size)
val_ds = df_to_dataset(val, shuffle=False, batch_size=batch_size)
test_ds = df_to_dataset(test,shuffle=False,batch_size=batch_size)

In [9]:
# Understand the input pipeline


for feature_batch, label_batch in train_ds.take(1):
    print('Every feature:', list(feature_batch.keys()))
    print('A batch of ages:', feature_batch['age'])
    print('A batch of targets:', label_batch)



Every feature: ['age', 'sex', 'cp', 'trestbps', 'chol', 'fbs', 'restecg', 'thalach', 'exang', 'oldpeak', 'slope', 'ca', 'thal']
A batch of ages: tf.Tensor([34 60 37 50 47], shape=(5,), dtype=int32)
A batch of targets: tf.Tensor([0 1 0 0 0], shape=(5,), dtype=int32)


In [10]:
# Demonstrate several types of feature column

# We will use this batch to demonstrate several types of
# feature columns
example_batch = next(iter(train_ds))[0]
example_batch

{'age': <tf.Tensor: id=90, shape=(5,), dtype=int32, numpy=array([34, 60, 37, 50, 47], dtype=int32)>,
 'sex': <tf.Tensor: id=98, shape=(5,), dtype=int32, numpy=array([0, 1, 0, 0, 1], dtype=int32)>,
 'cp': <tf.Tensor: id=93, shape=(5,), dtype=int32, numpy=array([2, 4, 3, 4, 3], dtype=int32)>,
 'trestbps': <tf.Tensor: id=102, shape=(5,), dtype=int32, numpy=array([118, 145, 120, 110, 108], dtype=int32)>,
 'chol': <tf.Tensor: id=92, shape=(5,), dtype=int32, numpy=array([210, 282, 215, 254, 243], dtype=int32)>,
 'fbs': <tf.Tensor: id=95, shape=(5,), dtype=int32, numpy=array([0, 0, 0, 0, 0], dtype=int32)>,
 'restecg': <tf.Tensor: id=97, shape=(5,), dtype=int32, numpy=array([0, 2, 0, 2, 0], dtype=int32)>,
 'thalach': <tf.Tensor: id=101, shape=(5,), dtype=int32, numpy=array([192, 142, 170, 159, 152], dtype=int32)>,
 'exang': <tf.Tensor: id=94, shape=(5,), dtype=int32, numpy=array([0, 1, 0, 0, 0], dtype=int32)>,
 'oldpeak': <tf.Tensor: id=96, shape=(5,), dtype=float32, numpy=array([0.7, 2.8, 0. 

In [20]:
# A utility method to create a feature column
# and to transform a batch of data
def demo(feature_colum):
    feature_layer = layers.DenseFeatures(feature_column)
    print('demo fn', feature_layer(example_batch).numpy())

In [21]:
# Numeric columns

age  = feature_column.numeric_column('age')
demo(age)

TypeError: 'DeprecationWrapper' object is not iterable

In [16]:
# Bucketized columns

age_buckets = feature_column.bucketized_column(age,
                                               boundaries=[18,25,30,35,40,45,50,55,60,65])
demo(age_buckets)

TypeError: 'DeprecationWrapper' object is not iterable

In [17]:
# Categorical columns

thal = feature_column.categorical_column_with_vocabulary_list(
    'thal',['fixed','normal','reversible']
)

thal_one_hot = feature_column.indicator_column(thal)
demo(thal_one_hot)

TypeError: 'DeprecationWrapper' object is not iterable

In [22]:
# Embedding columns

# Notice the input to the embedding column is the 
# categorical column we previously created
thal_embedding = feature_column.embedding_column(thal, dimension=8)
demo(thal_embedding)

TypeError: 'DeprecationWrapper' object is not iterable

In [23]:
#  Hashed feature columns

thal_hashed = feature_column.categorical_column_with_hash_bucket(
    'thal', hash_bucket_size=1000
)
demo(feature_column.indicator_column(thal_hashed))

TypeError: 'DeprecationWrapper' object is not iterable

In [24]:
# Crossed feature columns

crossed_feature = feature_column.crossed_column(
    [age_buckets, thal],
    hash_bucket_size=1000
)
demo(feature_column.indicator_column(crossed_feature))

TypeError: 'DeprecationWrapper' object is not iterable

In [25]:
# Choose which column to use

feature_columns = []

# numeric cols
for header in [
    'age','trestbps','chol','thalach','oldpeak','slope'
    'ca']:
    feature_columns.append(feature_column.numeric_column(header))

In [26]:
# bucketized cols
age_buckets = feature_column.bucketized_column(age,  boundaries=[18, 25, 30, 35, 40, 45, 50, 55, 60, 65])
feature_columns.append(age_buckets)

In [27]:
# indicator cols
thal = feature_column.categorical_column_with_vocabulary_list(
    'thal',['fixed','normal','reversible']
)
thal_one_hot = feature_column.indicator_column(thal)
feature_columns.append(thal_one_hot)

In [30]:
# embedding cols
thal_embedding = feature_column.embedding_column(thal,dimension=8)
feature_columns.append(thal_embedding)

In [31]:
# crossed cols
crossed_feature = feature_column.crossed_column([age_buckets,thal],
                                               hash_bucket_size=1000
                                               )
crossed_feature = feature_column.indicator_column(crossed_feature)
feature_columns.append(crossed_feature)

In [32]:
# Create a feature layer

feature_layer = tf.keras.layers.DenseFeatures(feature_columns)

In [33]:
batch_size = 32
train_ds = df_to_dataset(train,batch_size=batch_size)
val_ds = df_to_dataset(val, shuffle=False,batch_size=batch_size)
test_ds = df_to_dataset(test,shuffle=False, batch_size=batch_size)

In [34]:
# create, compile and train the model

model = tf.keras.Sequential([
    feature_layer,
    layers.Dense(128, activation='relu'),
    layers.Dense(128, activation='relu'),
    layers.Dense(1, activation='sigmoid')
])

model.compile(optimizer='adam',
             loss='binary_crossentropy',
             metrics=['accuracy'])

model.fit(train_ds,
         validation_data=val_ds,
         epochs=5)

Epoch 1/5
Instructions for updating:
The old _FeatureColumn APIs are being deprecated. Please use the new FeatureColumn APIs instead.
Instructions for updating:
Use tf.where in 2.0, which has the same broadcast rule as np.where
Instructions for updating:
The old _FeatureColumn APIs are being deprecated. Please use the new FeatureColumn APIs instead.
Instructions for updating:
The old _FeatureColumn APIs are being deprecated. Please use the new FeatureColumn APIs instead.


ValueError: in converted code:
    relative to /home/george/anaconda3/lib/python3.7/site-packages/tensorflow/python/feature_column:

    feature_column_v2.py:474 call
        self._state_manager)
    feature_column_v2.py:2799 get_dense_tensor
        return transformation_cache.get(self, state_manager)
    feature_column_v2.py:2562 get
        transformed = column.transform_feature(self, state_manager)
    feature_column_v2.py:2771 transform_feature
        input_tensor = transformation_cache.get(self.key, state_manager)
    feature_column_v2.py:2554 get
        raise ValueError('Feature {} is not in features dictionary.'.format(key))

    ValueError: Feature slopeca is not in features dictionary.


In [None]:
loss, accuracy = model.evaluate(test_ds)
print("Accuracy", accuracy)
