## A. 1 Hidden Layer Feedforward Neural Network (Sigmoid)

Steps:
1. Load dataset
2. Make dataset Iterable
3. Create model class
4. Instantiate model class
5. Instantiate loss class
6. Instantiate optimizer class
7. Train model

In [4]:
import torch
import torch.nn as nn
import torchvision.transforms as transforms
import torchvision.datasets as dsets
from torch.autograd import Variable

In [5]:
#Step 1: Load dataset

train_dataset = dsets.MNIST(root='./mnist', 
                            train=True, 
                            transform=transforms.ToTensor(),
                            download=True)

test_dataset = dsets.MNIST(root='./mnist', 
                           train=False, 
                           transform=transforms.ToTensor())

In [6]:
#Step 2: Make dataset iterable
batch_size = 100
n_iters = 3000
num_epochs = n_iters / (len(train_dataset) / batch_size)
num_epochs = int(num_epochs)

train_loader = torch.utils.data.DataLoader(dataset=train_dataset, 
                                           batch_size=batch_size, 
                                           shuffle=True)

test_loader = torch.utils.data.DataLoader(dataset=test_dataset, 
                                          batch_size=batch_size, 
                                          shuffle=False)


### Step 3: Create model class

In [46]:
class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function: 784 --> 100
        self.fc1 = nn.Linear(input_dim, hidden_dim) 
        
        # Non-linearity 
        self.sigmoid = nn.Sigmoid()
        
        # Linear function (readout):
        self.fc2 = nn.Linear(hidden_dim, output_dim)  
    
    def forward(self, x):
        # Linear function 1
        out = self.fc1(x)
        # Non-linearity 1
        out = self.sigmoid(out)
        # Linear function 4 (readout)
        out = self.fc2(out)
        return out

### Step 4: Instantiate model class
- Input dimension: 784
    - size of image
    - 28 x 28 = 784
- Output dimension: 10
    - 0,1,2,3,4,5,6,7,8,9
- Hidden dimension: 100
    - can be any number
    - also called: number of neurons, num of non-linear activation function

In [56]:
input_dim = 28*28
hidden_dim = 100
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim, hidden_dim, output_dim)

### Step 5: Instantiate Loss Class
- Feedforward neural network: Cross Entropy Loss
    - Logistic regression: Cross Entropy Loss
    - Linear Regression: MSE

In [57]:
criterion = nn.CrossEntropyLoss()

### Step 6: Instantiate optimizer class

In [58]:
learning_rate = 0.1

optimizer = torch.optim.SGD(model.parameters(), lr=learning_rate)

In [59]:
#Params
print(len(list(model.parameters())))

#Hidden layer params
print(list(model.parameters())[0].size())

#Bias params
print(list(model.parameters())[1].size())

#FC Param
print(list(model.parameters())[2].size())
#FC Bias
print(list(model.parameters())[3].size())

4
torch.Size([100, 784])
torch.Size([100])
torch.Size([10, 100])
torch.Size([10])


### Step 7: Train model


In [61]:
iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):

        images = Variable(images.view(-1, 28*28))
        labels = Variable(labels)
        
        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()
        
        # Forward pass to get output/logits
        outputs = model(images)
        
        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)
        
        # Getting gradients w.r.t. parameters
        loss.backward()
        
        # Updating parameters
        optimizer.step()
        
        iter += 1
        
        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                images = Variable(images.view(-1, 28*28))
                
                # Forward pass only to get logits/output
                outputs = model(images)
                
                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)
                
                # Total number of labels
                total += labels.size(0)
                
                # Total correct predictions
                correct += (predicted == labels).sum()
            
            accuracy = 100 * correct / total
            
            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.data[0], accuracy))

Iteration: 500. Loss: 0.36283618211746216. Accuracy: 89.48
Iteration: 1000. Loss: 0.5215944051742554. Accuracy: 90.48
Iteration: 1500. Loss: 0.4186626374721527. Accuracy: 91.31
Iteration: 2000. Loss: 0.3877609968185425. Accuracy: 91.61
Iteration: 2500. Loss: 0.2371458113193512. Accuracy: 92.1
Iteration: 3000. Loss: 0.2324649691581726. Accuracy: 92.31


## B. 1 Hidden Layer Feedforward Neural Network (ReLU)

In [63]:
class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function: 784 --> 100
        self.fc1 = nn.Linear(input_dim, hidden_dim) 
        
        # Non-linearity 
        self.relu = nn.ReLU()
        
        # Linear function (readout):
        self.fc2 = nn.Linear(hidden_dim, output_dim)  
    
    def forward(self, x):
        # Linear function 1
        out = self.fc1(x)
        # Non-linearity 1
        out = self.relu(out)
        # Linear function (readout)
        out = self.fc2(out)
        return out

In [64]:
#Train model(same as sigmoid)

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):

        images = Variable(images.view(-1, 28*28))
        labels = Variable(labels)
        
        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()
        
        # Forward pass to get output/logits
        outputs = model(images)
        
        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)
        
        # Getting gradients w.r.t. parameters
        loss.backward()
        
        # Updating parameters
        optimizer.step()
        
        iter += 1
        
        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                images = Variable(images.view(-1, 28*28))
                
                # Forward pass only to get logits/output
                outputs = model(images)
                
                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)
                
                # Total number of labels
                total += labels.size(0)
                
                # Total correct predictions
                correct += (predicted == labels).sum()
            
            accuracy = 100 * correct / total
            
            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.data[0], accuracy))

Iteration: 500. Loss: 0.29238009452819824. Accuracy: 92.51
Iteration: 1000. Loss: 0.305687814950943. Accuracy: 92.88
Iteration: 1500. Loss: 0.20709852874279022. Accuracy: 93.11
Iteration: 2000. Loss: 0.18495804071426392. Accuracy: 93.3
Iteration: 2500. Loss: 0.19227074086666107. Accuracy: 93.3
Iteration: 3000. Loss: 0.22882875800132751. Accuracy: 93.63
