In [199]:
import sklearn
import numpy as np
import h5py
import matplotlib.pyplot as plt
plt.rcParams['figure.figsize'] = (5.0, 4.0) # set default size of plots
plt.rcParams['image.interpolation'] = 'nearest'
plt.rcParams['image.cmap'] = 'gray'
np.random.seed(1)
from sklearn import datasets

In [207]:
digits = sklearn.datasets.load_digits()  # load image dataset for digits
X=digits.images   #image dataset from the dataset
Y=digits.target   #target data
Y=Y.reshape(1,X.shape[0])   
classes=digits.target_names    #output classes - 0 to 9 here
X_shape=X.shape
Y_shape=Y.shape
m=X_shape[0]   #trainining examples

X_train=X[:round(0.2*X_shape[0]),:] #rounded because the array dimensions must be an integer
X_test= X[round(0.2*X_shape[0]):,:]
Y_train=Y[:,:round(0.2*Y_shape[1])]
Y_test=Y[:,round(0.2*Y_shape[1]):]
m_train=X_train.shape[0]
m_test=X_test.shape[0]
num_px=X_train.shape[1]    # size of the image. here image size is 8x8. so num_px=8

#Reshape the training and test data sets so that images of size (num_px, num_px) are flattened into 
#single vectors of shape (num_px * num_px, 1)

X_train_flatten = X_train.reshape(X_train.shape[0], -1).T
X_test_flatten = X_test.reshape(X_test.shape[0], -1).T

#standardize the dataset

X_train_std = X_train_flatten / 255
X_test_std = X_test_flatten / 255


In [209]:
X_train_std.shape

(64, 359)

In [202]:
def sigmoid(z):
    """
    Compute the sigmoid of z

    Arguments:
    x -- A scalar or numpy array of any size.

    Return:
    s -- sigmoid(z)
    """

    ### START CODE HERE ### (≈ 1 line of code)
    s = 1 / (1 + np.exp(-z))
    ### END CODE HERE ###
    
    return s


In [203]:
def initialize_with_zeros(dim):
    """
    This function creates a vector of zeros of shape (dim, 1) for w and initializes b to 0.
    
    Argument:
    dim -- size of the w vector we want (or number of parameters in this case)
    
    Returns:
    w -- initialized vector of shape (dim, 1)
    b -- initialized scalar (corresponds to the bias)
    """
    
    ### START CODE HERE ### (≈ 1 line of code)
    w = np.zeros(shape=(dim, 1))
    b = 0
    ### END CODE HERE ###

    assert(w.shape == (dim, 1))
    assert(isinstance(b, float) or isinstance(b, int))
    
    return w, b

In [210]:
#w,b=initialize_with_zeros(X_train_std.shape[0])
#len(w)

In [205]:
def propagate(w, b, X, Y):
    """
    Implement the cost function and its gradient for the propagation explained above

    Arguments:
    w -- weights, a numpy array of size (num_px * num_px * 1, 1)
    b -- bias, a scalar
    X -- data of size (num_px * num_px * 1, number of examples)
    Y -- true "label" vector (containing 0 if false, 1 if true) of size (1, number of examples)

    Return:
    cost -- negative log-likelihood cost for logistic regression
    dw -- gradient of the loss with respect to w, thus same shape as w
    db -- gradient of the loss with respect to b, thus same shape as b
    
    Tips:
    - Write your code step by step for the propagation
    """
    
    m = X.shape[1]
    
    # FORWARD PROPAGATION (FROM X TO COST)
    
    A = sigmoid(np.dot(w.T, X) + b)  # compute activation
    cost = (- 1 / m) * np.sum(Y * np.log(A) + (1 - Y) * (np.log(1 - A)))  # compute cost
    
    
    # BACKWARD PROPAGATION (TO FIND GRAD)
    
    dw = (1 / m) * np.dot(X, (A - Y).T)
    db = (1 / m) * np.sum(A - Y)
    

    assert(dw.shape == w.shape)
    assert(db.dtype == float)
    cost = np.squeeze(cost)
    assert(cost.shape == ())
    
    grads = {"dw": dw,
             "db": db}
    
    return grads, cost

In [213]:
#w, b, X, Y = np.array([[1], [2]]), 2, np.array([[1,2], [3,4]]), np.array([[1, 0]])
(w,b),X,Y=initialize_with_zeros(X_train_std.shape[0]),X_train_std,Y_train
grads, cost = propagate(w, b, X, Y)
print ("dw = " + str(grads["dw"]))
print ("db = " + str(grads["db"]))
print ("cost = " + str(cost))

dw = [[ 0.00000000e+00]
 [-5.47271834e-03]
 [-8.29755858e-02]
 [-1.82549566e-01]
 [-1.78742695e-01]
 [-8.47506691e-02]
 [-1.65820088e-02]
 [-2.27210661e-03]
 [-4.91561527e-05]
 [-2.44961494e-02]
 [-1.65235676e-01]
 [-1.89081872e-01]
 [-1.57856792e-01]
 [-1.25719591e-01]
 [-2.68010268e-02]
 [-1.23436561e-03]
 [ 0.00000000e+00]
 [-3.14435524e-02]
 [-1.41607952e-01]
 [-1.13539789e-01]
 [-1.10098858e-01]
 [-1.27183352e-01]
 [-2.44852258e-02]
 [-1.80239227e-04]
 [-3.82325632e-05]
 [-2.44961494e-02]
 [-1.24621771e-01]
 [-1.57288765e-01]
 [-1.63285816e-01]
 [-1.23616800e-01]
 [-2.61619968e-02]
 [ 0.00000000e+00]
 [ 0.00000000e+00]
 [-2.14156972e-02]
 [-1.20356109e-01]
 [-1.65219291e-01]
 [-1.72925883e-01]
 [-1.35812988e-01]
 [-3.55835928e-02]
 [ 0.00000000e+00]
 [ 0.00000000e+00]
 [-1.52493309e-02]
 [-1.02638047e-01]
 [-1.24976787e-01]
 [-1.38631274e-01]
 [-1.19722541e-01]
 [-5.10677809e-02]
 [-1.20159484e-04]
 [ 0.00000000e+00]
 [-7.87044623e-03]
 [-1.01567535e-01]
 [-1.50707302e-01]
 [-1.41

In [225]:
def optimize(w, b, X, Y, num_iterations, learning_rate, print_cost = False):
    """
    This function optimizes w and b by running a gradient descent algorithm
    
    Arguments:
    w -- weights, a numpy array of size (num_px * num_px * 3, 1)
    b -- bias, a scalar
    X -- data of shape (num_px * num_px * 3, number of examples)
    Y -- true "label" vector (containing 0 if non-cat, 1 if cat), of shape (1, number of examples)
    num_iterations -- number of iterations of the optimization loop
    learning_rate -- learning rate of the gradient descent update rule
    print_cost -- True to print the loss every 100 steps
    
    Returns:
    params -- dictionary containing the weights w and bias b
    grads -- dictionary containing the gradients of the weights and bias with respect to the cost function
    costs -- list of all the costs computed during the optimization, this will be used to plot the learning curve.
    
    Tips:
    You basically need to write down two steps and iterate through them:
        1) Calculate the cost and the gradient for the current parameters. Use propagate().
        2) Update the parameters using gradient descent rule for w and b.
    """
    
    costs = []
    
    for i in range(num_iterations):
        
        
        # Cost and gradient calculation (≈ 1-4 lines of code)
        ### START CODE HERE ### 
        grads, cost = propagate(w, b, X, Y)
        ### END CODE HERE ###
        
        # Retrieve derivatives from grads
        dw = grads["dw"]
        db = grads["db"]
        
        # update rule (≈ 2 lines of code)
        ### START CODE HERE ###
        w = w - learning_rate * dw  # need to broadcast
        b = b - learning_rate * db
        ### END CODE HERE ###
        
        # Record the costs
        if i % 100 == 0:
            costs.append(cost)
        
        # Print the cost every 100 training examples
        if print_cost and i % 100 == 0:
            print ("Cost after iteration %i: %f" % (i, cost))
    
    params = {"w": w,
              "b": b}
    
    grads = {"dw": dw,
             "db": db}
    
    return params, grads, costs

In [228]:
params, grads, costs = optimize(w, b, X, Y, num_iterations= 1000, learning_rate = 0.009, print_cost = True)
print ("w = " + str(params["w"]))
#print ("b = " + str(params["b"]))
#print ("dw = " + str(grads["dw"]))
#print ("db = " + str(grads["db"]))

Cost after iteration 0: 0.693147
Cost after iteration 100: -11.410239
Cost after iteration 200: -22.305032
Cost after iteration 300: -33.137169
Cost after iteration 400: -43.966689
Cost after iteration 500: -54.796101
Cost after iteration 600: -65.625508
Cost after iteration 700: -76.454915
Cost after iteration 800: -87.284324
Cost after iteration 900: -98.113766
w = [[0.00000000e+00]
 [4.26530330e-02]
 [6.59357676e-01]
 [1.45286900e+00]
 [1.40988463e+00]
 [6.68215250e-01]
 [1.32923802e-01]
 [1.89401155e-02]
 [3.95249541e-04]
 [1.92502662e-01]
 [1.32090365e+00]
 [1.49911695e+00]
 [1.22643707e+00]
 [9.85865958e-01]
 [2.11362210e-01]
 [1.02605782e-02]
 [0.00000000e+00]
 [2.46352871e-01]
 [1.12371987e+00]
 [8.90674860e-01]
 [8.56499989e-01]
 [1.00629971e+00]
 [1.91839507e-01]
 [1.48069537e-03]
 [2.96934567e-04]
 [1.83638849e-01]
 [9.75278919e-01]
 [1.25914033e+00]
 [1.30321315e+00]
 [9.84247402e-01]
 [2.01602096e-01]
 [0.00000000e+00]
 [0.00000000e+00]
 [1.58178013e-01]
 [9.42498272e-01]


In [192]:
def predict(w, b, X):
    '''
    Predict whether the label is 0 or 1 using learned logistic regression parameters (w, b)
    
    Arguments:
    w -- weights, a numpy array of size (num_px * num_px , 1)
    b -- bias, a scalar
    X -- data of size (num_px * num_px , number of examples)
    
    Returns:
    Y_prediction -- a numpy array (vector) containing all predictions (0/1) for the examples in X
    '''
    
    m = X.shape[1]
    Y_prediction = np.zeros((1, m))
    w = w.reshape(X.shape[0], 1)
    
    # Compute vector "A" predicting the probabilities of a cat being present in the picture
   
    A = sigmoid(np.dot(w.T, X) + b)
    
    
    for i in range(A.shape[1]):
        # Convert probabilities a[0,i] to actual predictions p[0,i]
       
        Y_prediction[0, i] = 1 if A[0, i] > 0.5 else 0
       
    
    assert(Y_prediction.shape == (1, m))
    
    return Y_prediction

In [193]:
#print("predictions = " + str(predict(w, b, X)))

In [194]:
def model(X_train, Y_train, X_test, Y_test, num_iterations=2000, learning_rate=0.5, print_cost=False):
    """
    Builds the logistic regression model by calling the function you've implemented previously
    
    Arguments:
    X_train -- training set represented by a numpy array of shape (num_px * num_px * 3, m_train)
    Y_train -- training labels represented by a numpy array (vector) of shape (1, m_train)
    X_test -- test set represented by a numpy array of shape (num_px * num_px * 3, m_test)
    Y_test -- test labels represented by a numpy array (vector) of shape (1, m_test)
    num_iterations -- hyperparameter representing the number of iterations to optimize the parameters
    learning_rate -- hyperparameter representing the learning rate used in the update rule of optimize()
    print_cost -- Set to true to print the cost every 100 iterations
    
    Returns:
    d -- dictionary containing information about the model.
    """
    
   
    # initialize parameters with zeros 
    w, b = initialize_with_zeros(X_train.shape[0])

    # Gradient descent 
    parameters, grads, costs = optimize(w, b, X_train, Y_train, num_iterations, learning_rate, print_cost)
    
    # Retrieve parameters w and b from dictionary "parameters"
    w = parameters["w"]
    b = parameters["b"]
    
    # Predict test/train set examples 
    Y_prediction_test = predict(w, b, X_test)
    Y_prediction_train = predict(w, b, X_train)

  

    # Print train/test Errors
    print("train accuracy: {} %".format(100 - np.mean(np.abs(Y_prediction_train - Y_train)) * 100))
    print("test accuracy: {} %".format(100 - np.mean(np.abs(Y_prediction_test - Y_test)) * 100))

    
    d = {"costs": costs,
         "Y_prediction_test": Y_prediction_test, 
         "Y_prediction_train" : Y_prediction_train, 
         "w" : w, 
         "b" : b,
         "learning_rate" : learning_rate,
         "num_iterations": num_iterations}
    
    return d

In [195]:
X_train.shape

(359, 8, 8)

In [196]:
d = model(X_train_std, Y_train, X_test_std, Y_test, num_iterations = 2000, learning_rate = 0.005, print_cost = True)

Cost after iteration 0: 0.693147
Cost after iteration 100: -6.389894
Cost after iteration 200: -12.630515
Cost after iteration 300: -18.686551
Cost after iteration 400: -24.709697
Cost after iteration 500: -30.727196
Cost after iteration 600: -36.743731
Cost after iteration 700: -42.760102
Cost after iteration 800: -48.776445
Cost after iteration 900: -54.792783
Cost after iteration 1000: -60.809120
Cost after iteration 1100: -66.825458
Cost after iteration 1200: -72.841795
Cost after iteration 1300: -78.858132
Cost after iteration 1400: -84.874469
Cost after iteration 1500: -90.890807
Cost after iteration 1600: -96.907179
Cost after iteration 1700: -102.923275
Cost after iteration 1800: -108.940754
Cost after iteration 1900: -114.962218
train accuracy: -261.00278551532034 %
test accuracy: -270.8623087621697 %


In [68]:

# Example of a picture that was wrongly classified.# Exampl 
index = 5
plt.imshow(test_set_x[:,index].reshape((num_px, num_px, 3)))
print ("y = " + str(test_set_y[0, index]) + ", you predicted that it is a \"" + classes[d["Y_prediction_test"][0, index]].decode("utf-8") +  "\" picture.")

NameError: name 'test_set_x' is not defined

In [69]:
learning_rates = [0.01, 0.001, 0.0001]
models = {}
for i in learning_rates:
    print ("learning rate is: " + str(i))
    models[str(i)] = model(train_set_x, train_set_y, test_set_x, test_set_y, num_iterations = 1500, learning_rate = i, print_cost = False)
    print ('\n' + "-------------------------------------------------------" + '\n')

for i in learning_rates:
    plt.plot(np.squeeze(models[str(i)]["costs"]), label= str(models[str(i)]["learning_rate"]))

plt.ylabel('cost')
plt.xlabel('iterations')

legend = plt.legend(loc='upper center', shadow=True)
frame = legend.get_frame()
frame.set_facecolor('0.90')
plt.show()

learning rate is: 0.01


NameError: name 'train_set_x' is not defined

In [66]:
#d = model(train_set_x, train_set_y, test_set_x, test_set_y, num_iterations = 2000, learning_rate = 0.005, print_cost = True)

In [67]:
## START CODE HERE ## (PUT YOUR IMAGE NAME) 
my_image = "my_image.jpg"   # change this to the name of your image file 
## END CODE HERE ##

# We preprocess the image to fit your algorithm.
fname = "images/" + my_image
image = np.array(ndimage.imread(fname, flatten=False))
my_image = scipy.misc.imresize(image, size=(num_px, num_px)).reshape((1, num_px * num_px * 3)).T
my_predicted_image = predict(d["w"], d["b"], my_image)

plt.imshow(image)
print("y = " + str(np.squeeze(my_predicted_image)) + ", your algorithm predicts a \"" + classes[int(np.squeeze(my_predicted_image)),].decode("utf-8") +  "\" picture.")

NameError: name 'ndimage' is not defined