In [0]:
## Step 1: Loading MNIST Train Dataset

import torch
import torch.nn as nn
import torchvision.transforms as transforms
import torchvision.datasets as dsets

train_dataset = dsets.MNIST(root='./data', 
                            train=True, 
                            transform=transforms.ToTensor(),
                            download=True)

test_dataset = dsets.MNIST(root='./data', 
                           train=False, 
                           transform=transforms.ToTensor())

## Step 2: Make Dataset Iterable

batch_size = 100
n_iters = 6000
num_epochs = n_iters / (len(train_dataset) / batch_size)
num_epochs = int(num_epochs)

train_loader = torch.utils.data.DataLoader(dataset=train_dataset, 
                                           batch_size=batch_size, 
                                           shuffle=True)

test_loader = torch.utils.data.DataLoader(dataset=test_dataset, 
                                          batch_size=batch_size, 
                                          shuffle=False)

**MODEL_1_onehiddenlayer_Sigmoid**


In [42]:
## Step 3: Create Model Class

class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function
        self.fc1 = nn.Linear(input_dim, hidden_dim) 

        # Non-linearity
        self.sigmoid = nn.Sigmoid()

        # Linear function (readout)
        self.fc2 = nn.Linear(hidden_dim, output_dim)  

    def forward(self, x):
        # Linear function  # LINEAR
        out = self.fc1(x)

        # Non-linearity  # NON-LINEAR
        out = self.sigmoid(out)

        # Linear function (readout)  # LINEAR
        out = self.fc2(out)
        return out


## Step 4: Instantiate Model Class

input_dim = 28*28
hidden_dim = 64
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim, hidden_dim, output_dim)

## Step 5: Instantiate Loss Class
criterion = nn.CrossEntropyLoss()

##STEP 6: INSTANTIATE OPTIMIZER CLASS
#optimizer = torch.optim.SGD(model.parameters(), lr=0.01)
optimizer=torch.optim.Adam(model.parameters(),lr=0.01)

## Step 7: Train Model

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 600 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            parameters=sum(p.numel() for p in model.parameters())
            print("Parameters:",parameters)
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Parameters: 50890
Iteration: 600. Loss: 0.19823502004146576. Accuracy: 95
Parameters: 50890
Iteration: 1200. Loss: 0.1227787509560585. Accuracy: 96
Parameters: 50890
Iteration: 1800. Loss: 0.16046862304210663. Accuracy: 96
Parameters: 50890
Iteration: 2400. Loss: 0.09352443367242813. Accuracy: 96
Parameters: 50890
Iteration: 3000. Loss: 0.02999391034245491. Accuracy: 96
Parameters: 50890
Iteration: 3600. Loss: 0.04375356063246727. Accuracy: 96
Parameters: 50890
Iteration: 4200. Loss: 0.04025638476014137. Accuracy: 97
Parameters: 50890
Iteration: 4800. Loss: 0.05508409067988396. Accuracy: 96
Parameters: 50890
Iteration: 5400. Loss: 0.14829429984092712. Accuracy: 96
Parameters: 50890
Iteration: 6000. Loss: 0.049088068306446075. Accuracy: 96


**MODEL_2_singlehiddenlayer_ReLU**

In [31]:
## Step 3: Create Model Class

class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function
        self.fc1 = nn.Linear(input_dim, hidden_dim) 

        # Non-linearity
        self.relu = nn.ReLU()

        # Linear function (readout)
        self.fc2 = nn.Linear(hidden_dim, output_dim)  

    def forward(self, x):
        # Linear function  # LINEAR
        out = self.fc1(x)

        # Non-linearity  # NON-LINEAR
        out = self.relu(out)

        # Linear function (readout)  # LINEAR
        out = self.fc2(out)
        return out


## Step 4: Instantiate Model Class

input_dim = 28*28
hidden_dim = 64
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim, hidden_dim, output_dim)

## Step 5: Instantiate Loss Class
criterion = nn.CrossEntropyLoss()

##STEP 6: INSTANTIATE OPTIMIZER CLASS
#optimizer = torch.optim.SGD(model.parameters(), lr=0.01)
optimizer=torch.optim.Adam(model.parameters(),lr=0.01)

## Step 7: Train Model

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 600 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            parameters=sum(p.numel() for p in model.parameters())
            print("Parameters:",parameters)
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Iteration: 600. Loss: 0.10354761779308319. Accuracy: 95
Iteration: 1200. Loss: 0.14353252947330475. Accuracy: 96
Iteration: 1800. Loss: 0.11414793878793716. Accuracy: 96
Iteration: 2400. Loss: 0.05229035019874573. Accuracy: 96
Iteration: 3000. Loss: 0.08955443650484085. Accuracy: 96


**Model_3_2hiddenlayers_ReLU**

In [6]:
## Step 3: Create Model Class

class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim1,hidden_dim12, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function 1 (784 --> 100)
        self.fc1 = nn.Linear(input_dim, hidden_dim1) 
        # Non-linearity 1
        self.relu1 = nn.ReLU()

        # Linear Function 2: (100 --> 100)
        self.fc2 = nn.Linear(hidden_dim1,hidden_dim2)
        # Non-Linearity 2
        self.relu2 = nn.ReLU()

        # Linear function (readout)
        self.fc3 = nn.Linear(hidden_dim2, output_dim)  

    def forward(self, x):
        # Linear function  # LINEAR 1
        out = self.fc1(x)
        # Non-linearity  # NON-LINEAR 1
        out = self.relu1(out)

        #Linear Function 2
        out = self.fc2(out)
        # Non - linearity 2
        out = self.relu2(out)

        # Linear function (readout)  # LINEAR
        out = self.fc3(out)
        return out


## Step 4: Instantiate Model Class

input_dim = 28*28
hidden_dim1 = 64
hidden_dim2= 32
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim,hidden_dim1,hidden_dim2, output_dim)

## Step 5: Instantiate Loss Class
criterion = nn.CrossEntropyLoss()

##STEP 6: INSTANTIATE OPTIMIZER CLASS
learning_rate = 0.2
optimizer = torch.optim.SGD(model.parameters(), lr=1)

## Step 7: Train Model

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = float(100 * correct / total)

            # Print Loss
            parameters=sum(p.numel() for p in model.parameters())
            print("Parameters:",parameters)
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Iteration: 500. Loss: 0.5825788378715515. Accuracy: 86.0
Iteration: 1000. Loss: 0.2996422052383423. Accuracy: 92.0
Iteration: 1500. Loss: 0.2911902964115143. Accuracy: 93.0
Iteration: 2000. Loss: 0.13938476145267487. Accuracy: 93.0
Iteration: 2500. Loss: 0.2750631868839264. Accuracy: 93.0
Iteration: 3000. Loss: 0.22518295049667358. Accuracy: 94.0


**Model_4_2hiddenlayers_Sigmoid**

In [43]:
## Step 3: Create Model Class

class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim1,hidden_dim12, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function 1 (784 --> 100)
        self.fc1 = nn.Linear(input_dim, hidden_dim1) 
        # Non-linearity 1
        self.sigmoid1 = nn.Sigmoid()

        # Linear Function 2: (100 --> 100)
        self.fc2 = nn.Linear(hidden_dim1,hidden_dim2)
        # Non-Linearity 2
        self.sigmoid2 = nn.Sigmoid()

        # Linear function (readout)
        self.fc3 = nn.Linear(hidden_dim2, output_dim)  

    def forward(self, x):
        # Linear function  # LINEAR 1
        out = self.fc1(x)
        # Non-linearity  # NON-LINEAR 1
        out = self.sigmoid1(out)

        #Linear Function 2
        out = self.fc2(out)
        # Non - linearity 2
        out = self.sigmoid2(out)

        # Linear function (readout)  # LINEAR
        out = self.fc3(out)
        return out


## Step 4: Instantiate Model Class

input_dim = 28*28
hidden_dim1 = 64
hidden_dim2= 32
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim,hidden_dim1,hidden_dim2, output_dim)

## Step 5: Instantiate Loss Class
criterion = nn.CrossEntropyLoss()

##STEP 6: INSTANTIATE OPTIMIZER CLASS
optimizer = torch.optim.SGD(model.parameters(), lr=0.1)

## Step 7: Train Model

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = float(100 * correct / total)

            # Print Loss
            parameters=sum(p.numel() for p in model.parameters())
            print("Parameters:",parameters)
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Parameters: 52650
Iteration: 500. Loss: 2.2086191177368164. Accuracy: 50.0
Parameters: 52650
Iteration: 1000. Loss: 1.2478301525115967. Accuracy: 63.0
Parameters: 52650
Iteration: 1500. Loss: 0.7558260560035706. Accuracy: 80.0
Parameters: 52650
Iteration: 2000. Loss: 0.4983283281326294. Accuracy: 85.0
Parameters: 52650
Iteration: 2500. Loss: 0.5372934341430664. Accuracy: 87.0
Parameters: 52650
Iteration: 3000. Loss: 0.40647831559181213. Accuracy: 89.0
Parameters: 52650
Iteration: 3500. Loss: 0.33674943447113037. Accuracy: 89.0
Parameters: 52650
Iteration: 4000. Loss: 0.3637329936027527. Accuracy: 90.0
Parameters: 52650
Iteration: 4500. Loss: 0.29772165417671204. Accuracy: 90.0
Parameters: 52650
Iteration: 5000. Loss: 0.3259194493293762. Accuracy: 91.0
Parameters: 52650
Iteration: 5500. Loss: 0.19395916163921356. Accuracy: 91.0
Parameters: 52650
Iteration: 6000. Loss: 0.31262481212615967. Accuracy: 92.0


**Model_4_3hiddenlayers_ReLU**

In [22]:
## Step 3: Create Model Class

class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim1,hidden_dim2,hidden_dim3, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function 1
        self.fc1 = nn.Linear(input_dim, hidden_dim1) 
        # Non-linearity 1
        self.relu1 = nn.ReLU()

        # Linear function 2
        self.fc2 = nn.Linear(hidden_dim1, hidden_dim2)
        # Non-linearity 2
        self.relu2 = nn.ReLU()

        # Linear function 3
        self.fc3 = nn.Linear(hidden_dim2, hidden_dim3)
        # Non-linearity 3
        self.relu3 = nn.ReLU()

        # Linear function 4 (readout):
        self.fc4 = nn.Linear(hidden_dim3, output_dim)

    def forward(self, x):
        # Linear function 1
        out = self.fc1(x)
        # Non-linearity 1
        out = self.relu1(out)

        # Linear function 2
        out = self.fc2(out)
        # Non-linearity 2
        out = self.relu2(out)

        # Linear function 2
        out = self.fc3(out)
        # Non-linearity 2
        out = self.relu3(out)

        # Linear function 4 (readout)
        out = self.fc4(out)
        return out# Linear function  # LINEAR 1
        out = self.fc1(x)
        # Non-linearity  # NON-LINEAR 1
        out = self.relu1(out)

        #Linear Function 2
        out = self.fc2(out)
        # Non - linearity 2
        out = self.relu2(out)

        # Linear function (readout)  # LINEAR
        out = self.fc3(out)
        return out


## Step 4: Instantiate Model Class

input_dim = 28*28
hidden_dim1 = 64
hidden_dim2= 32
hidden_dim3= 16
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim, hidden_dim1,hidden_dim2,hidden_dim3, output_dim)

## Step 5: Instantiate Loss Class
criterion = nn.CrossEntropyLoss()

##STEP 6: INSTANTIATE OPTIMIZER CLASS
optimizer = torch.optim.SGD(model.parameters(), lr=0.1)

## Step 7: Train Model

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 600 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            parameters=sum(p.numel() for p in model.parameters())
            print("Parameters:",parameters)
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Parameters: 53018
Iteration: 600. Loss: 0.40132859349250793. Accuracy: 90
Parameters: 53018
Iteration: 1200. Loss: 0.24722160398960114. Accuracy: 92
Parameters: 53018
Iteration: 1800. Loss: 0.06277034431695938. Accuracy: 95
Parameters: 53018
Iteration: 2400. Loss: 0.09136150032281876. Accuracy: 96
Parameters: 53018
Iteration: 3000. Loss: 0.14660100638866425. Accuracy: 96
Parameters: 53018
Iteration: 3600. Loss: 0.22932130098342896. Accuracy: 96
Parameters: 53018
Iteration: 4200. Loss: 0.07837072759866714. Accuracy: 96
Parameters: 53018
Iteration: 4800. Loss: 0.050865963101387024. Accuracy: 96
Parameters: 53018
Iteration: 5400. Loss: 0.07254394888877869. Accuracy: 96
Parameters: 53018
Iteration: 6000. Loss: 0.03610512986779213. Accuracy: 96


**Model_5_3hiddenlayers_Sigmoid**

In [23]:
## Step 3: Create Model Class

class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim1,hidden_dim2,hidden_dim3, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function 1
        self.fc1 = nn.Linear(input_dim, hidden_dim1) 
        # Non-linearity 1
        self.sigmoid1 = nn.Sigmoid()

        # Linear function 2
        self.fc2 = nn.Linear(hidden_dim1, hidden_dim2)
        # Non-linearity 2
        self.sigmoid2 = nn.Sigmoid()

        # Linear function 3
        self.fc3 = nn.Linear(hidden_dim2, hidden_dim3)
        # Non-linearity 3
        self.sigmoid3 = nn.Sigmoid()

        # Linear function 4 (readout):
        self.fc4 = nn.Linear(hidden_dim3, output_dim)

    def forward(self, x):
        # Linear function 1
        out = self.fc1(x)
        # Non-linearity 1
        out = self.sigmoid1(out)

        # Linear function 2
        out = self.fc2(out)
        # Non-linearity 2
        out = self.sigmoid2(out)

        # Linear function 2
        out = self.fc3(out)
        # Non-linearity 2
        out = self.sigmoid3(out)

        # Linear function 4 (readout)
        out = self.fc4(out)
        return out# Linear function  # LINEAR 1
        out = self.fc1(x)
        # Non-linearity  # NON-LINEAR 1
        out = self.sigmoid1(out)

        #Linear Function 2
        out = self.fc2(out)
        # Non - linearity 2
        out = self.sigmoid2(out)

        # Linear function (readout)  # LINEAR
        out = self.fc3(out)
        return out


## Step 4: Instantiate Model Class

input_dim = 28*28
hidden_dim1 = 64
hidden_dim2= 32
hidden_dim3= 16
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim, hidden_dim1,hidden_dim2,hidden_dim3, output_dim)

## Step 5: Instantiate Loss Class
criterion = nn.CrossEntropyLoss()

##STEP 6: INSTANTIATE OPTIMIZER CLASS
optimizer = torch.optim.SGD(model.parameters(), lr=0.1)

## Step 7: Train Model

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 600 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            parameters=sum(p.numel() for p in model.parameters())
            print("Parameters:",parameters)
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Parameters: 53018
Iteration: 600. Loss: 2.3088741302490234. Accuracy: 11
Parameters: 53018
Iteration: 1200. Loss: 2.30619215965271. Accuracy: 11
Parameters: 53018
Iteration: 1800. Loss: 2.295761823654175. Accuracy: 11
Parameters: 53018
Iteration: 2400. Loss: 2.285662889480591. Accuracy: 11
Parameters: 53018
Iteration: 3000. Loss: 2.2355639934539795. Accuracy: 21
Parameters: 53018
Iteration: 3600. Loss: 1.4160528182983398. Accuracy: 46
Parameters: 53018
Iteration: 4200. Loss: 1.2083994150161743. Accuracy: 56
Parameters: 53018
Iteration: 4800. Loss: 1.04866361618042. Accuracy: 64
Parameters: 53018
Iteration: 5400. Loss: 0.8724411129951477. Accuracy: 78
Parameters: 53018
Iteration: 6000. Loss: 0.5630941987037659. Accuracy: 84


**Model_6_4hiddenlayers_Sigmoid**

In [44]:
## Step 3: Create Model Class

class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim1,hidden_dim2,hidden_dim3,hidden_dim4, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function 1
        self.fc1 = nn.Linear(input_dim, hidden_dim1)
        self.fc2 = nn.Linear(hidden_dim1, hidden_dim2)
        self.fc3 = nn.Linear(hidden_dim2, hidden_dim3)
        self.fc4 = nn.Linear(hidden_dim3, hidden_dim4)
        self.fc5 = nn.Linear(hidden_dim4, output_dim)
        # Non-linearity 1
        self.sigmoid1 = nn.Sigmoid()
        self.sigmoid2 = nn.Sigmoid()
        self.sigmoid3 = nn.Sigmoid()
        self.sigmoid4 = nn.Sigmoid()

    def forward(self, x):
        # Linear function 1
        out = self.fc1(x)
        out = self.fc2(out)
        out = self.fc3(out)
        out = self.fc4(out)
        out = self.fc5(out)

        # Non-linearity 1
        out = self.sigmoid1(out)
        out = self.sigmoid2(out)
        out = self.sigmoid3(out)
        out = self.sigmoid4(out)
      
        return out

## Step 4: Instantiate Model Class

input_dim = 28*28
hidden_dim1 = 128
hidden_dim2 = 64
hidden_dim3 = 32
hidden_dim4 = 16
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim, hidden_dim1,hidden_dim2,hidden_dim3,hidden_dim4, output_dim)

## Step 5: Instantiate Loss Class
criterion = nn.CrossEntropyLoss()

##STEP 6: INSTANTIATE OPTIMIZER CLASS
optimizer = torch.optim.SGD(model.parameters(), lr=0.1)

## Step 7: Train Model

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 600 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            parameters=sum(p.numel() for p in model.parameters())
            print("Parameters:",parameters)
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Parameters: 111514
Iteration: 600. Loss: 2.3025522232055664. Accuracy: 9
Parameters: 111514
Iteration: 1200. Loss: 2.302579879760742. Accuracy: 10
Parameters: 111514
Iteration: 1800. Loss: 2.302558422088623. Accuracy: 11
Parameters: 111514
Iteration: 2400. Loss: 2.302553415298462. Accuracy: 13
Parameters: 111514
Iteration: 3000. Loss: 2.3024749755859375. Accuracy: 13
Parameters: 111514
Iteration: 3600. Loss: 2.3025853633880615. Accuracy: 14
Parameters: 111514
Iteration: 4200. Loss: 2.302595615386963. Accuracy: 15
Parameters: 111514
Iteration: 4800. Loss: 2.302501678466797. Accuracy: 15
Parameters: 111514
Iteration: 5400. Loss: 2.302471399307251. Accuracy: 16
Parameters: 111514
Iteration: 6000. Loss: 2.3024606704711914. Accuracy: 16


**Model_7_4hiddenlayers_ReLU**

In [41]:
## Step 3: Create Model Class

class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim1,hidden_dim2,hidden_dim3,hidden_dim4, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function 1
        self.fc1 = nn.Linear(input_dim, hidden_dim1)
        self.fc2 = nn.Linear(hidden_dim1, hidden_dim2)
        self.fc3 = nn.Linear(hidden_dim2, hidden_dim3)
        self.fc4 = nn.Linear(hidden_dim3, hidden_dim4)
        self.fc5 = nn.Linear(hidden_dim4, output_dim)
        # Non-linearity 1
        self.relu1 = nn.ReLU()
        self.relu2 = nn.ReLU()
        self.relu3 = nn.ReLU()
        self.relu4 = nn.ReLU()

    def forward(self, x):
        # Linear function 1
        out = self.fc1(x)
        out = self.fc2(out)
        out = self.fc3(out)
        out = self.fc4(out)
        out = self.fc5(out)

        # Non-linearity 1
        out = self.relu1(out)
        out = self.relu2(out)
        out = self.relu3(out)
        out = self.relu4(out)
      
        return out#


## Step 4: Instantiate Model Class

input_dim = 28*28
hidden_dim1 = 128
hidden_dim2 = 64
hidden_dim3 = 32
hidden_dim4 = 16
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim, hidden_dim1,hidden_dim2,hidden_dim3,hidden_dim4, output_dim)

## Step 5: Instantiate Loss Class
criterion = nn.CrossEntropyLoss()

##STEP 6: INSTANTIATE OPTIMIZER CLASS
optimizer = torch.optim.SGD(model.parameters(), lr=0.02,momentum=.9)

## Step 7: Train Model

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 600 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            parameters=sum(p.numel() for p in model.parameters())
            print("Parameters:",parameters)
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Parameters: 111514
Iteration: 600. Loss: 0.5022249817848206. Accuracy: 89
Parameters: 111514
Iteration: 1200. Loss: 0.17718960344791412. Accuracy: 90
Parameters: 111514
Iteration: 1800. Loss: 0.3479868769645691. Accuracy: 90
Parameters: 111514
Iteration: 2400. Loss: 0.3205270767211914. Accuracy: 91
Parameters: 111514
Iteration: 3000. Loss: 0.25322726368904114. Accuracy: 91
Parameters: 111514
Iteration: 3600. Loss: 0.22225545346736908. Accuracy: 91
Parameters: 111514
Iteration: 4200. Loss: 0.6856802105903625. Accuracy: 91
Parameters: 111514
Iteration: 4800. Loss: 0.34979239106178284. Accuracy: 91
Parameters: 111514
Iteration: 5400. Loss: 0.35401150584220886. Accuracy: 91
Parameters: 111514
Iteration: 6000. Loss: 0.2134648561477661. Accuracy: 90
