<a href="https://colab.research.google.com/github/DuncanJF/dsml_jupyter/blob/main/workspace/CNN%2BMNIST%2Bpytorch.ipynb" target="_parent"><img src="https://colab.research.google.com/assets/colab-badge.svg" alt="Open In Colab"/></a>

In [None]:
import torch
import torch.nn as nn
import torchvision.transforms as transforms
import torchvision.datasets as dsets

In [None]:
#STEP 1: LOADING DATASET
train_dataset = dsets.MNIST(root='./data',
                            train=True,
                            transform=transforms.ToTensor(),
                            download=True)

In [None]:
print(train_dataset)

Dataset MNIST
    Number of datapoints: 60000
    Root location: ./data
    Split: Train
    StandardTransform
Transform: ToTensor()


In [None]:
test_dataset = dsets.MNIST(root='./data',
                           train=False,
                           transform=transforms.ToTensor())

In [None]:
print(test_dataset)

Dataset MNIST
    Number of datapoints: 10000
    Root location: ./data
    Split: Test
    StandardTransform
Transform: ToTensor()


In [None]:
#STEP 2: MAKING DATASET ITERABLE
batch_size = 100
n_iters = 3000
num_epochs = n_iters / (len(train_dataset) / batch_size)
num_epochs = int(num_epochs)

train_loader = torch.utils.data.DataLoader(dataset=train_dataset,
                                           batch_size=batch_size,
                                           shuffle=True)

test_loader = torch.utils.data.DataLoader(dataset=test_dataset,
                                          batch_size=batch_size,
                                          shuffle=False)

In [None]:
#STEP 3: CREATE MODEL CLASS
class FeedforwardNeuralNetModel(nn.Module):
    def __init__(self, input_dim, hidden_dim, output_dim):
        super(FeedforwardNeuralNetModel, self).__init__()
        # Linear function 1: 784 --> 100
        self.fc1 = nn.Linear(input_dim, hidden_dim)
        # Non-linearity 1
        self.relu1 = nn.ReLU()

        # Linear function 2: 100 --> 100
        self.fc2 = nn.Linear(hidden_dim, hidden_dim)
        # Non-linearity 2
        self.relu2 = nn.ReLU()

        # Linear function 3: 100 --> 100
        self.fc3 = nn.Linear(hidden_dim, hidden_dim)
        # Non-linearity 3
        self.relu3 = nn.ReLU()

        # Linear function 4 (readout): 100 --> 10
        self.fc4 = nn.Linear(hidden_dim, output_dim)

    def forward(self, x):
        # Linear function 1
        out = self.fc1(x)
        # Non-linearity 1
        out = self.relu1(out)

        # Linear function 2
        out = self.fc2(out)
        # Non-linearity 2
        out = self.relu2(out)

        # Linear function 2
        out = self.fc3(out)
        # Non-linearity 2
        out = self.relu3(out)

        # Linear function 4 (readout)
        out = self.fc4(out)
        return out

In [None]:
#STEP 4: INSTANTIATE MODEL CLASS
input_dim = 28*28
hidden_dim = 100
output_dim = 10

model = FeedforwardNeuralNetModel(input_dim, hidden_dim, output_dim)

In [None]:
#STEP 5: INSTANTIATE LOSS CLASS
criterion = nn.CrossEntropyLoss()

In [None]:
#STEP 6: INSTANTIATE OPTIMIZER CLASS
learning_rate = 0.1

optimizer = torch.optim.SGD(model.parameters(), lr=learning_rate)

In [None]:
print(num_epochs)

13


In [None]:
#STEP 7: TRAIN THE MODEL
iter = 0
iteration_prints=[1,2,4,8,16,32,64,128,256,512,1024,2048,4096,8192]
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        #if iter % 500 == 0:
        if iter in iteration_prints:
            # Calculate Accuracy
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))


Iteration: 1. Loss: 2.301947593688965. Accuracy: 9.739999771118164
Iteration: 2. Loss: 2.297227382659912. Accuracy: 9.739999771118164
Iteration: 4. Loss: 2.3088860511779785. Accuracy: 9.739999771118164
Iteration: 8. Loss: 2.297534942626953. Accuracy: 9.789999961853027
Iteration: 16. Loss: 2.294639825820923. Accuracy: 18.799999237060547
Iteration: 32. Loss: 2.274282455444336. Accuracy: 26.209999084472656
Iteration: 64. Loss: 2.232170343399048. Accuracy: 41.400001525878906
Iteration: 128. Loss: 1.244429588317871. Accuracy: 58.599998474121094
Iteration: 256. Loss: 0.5970173478126526. Accuracy: 84.77999877929688
Iteration: 512. Loss: 0.3231336176395416. Accuracy: 90.33999633789062
Iteration: 1024. Loss: 0.09996500611305237. Accuracy: 93.7699966430664
Iteration: 2048. Loss: 0.06200489029288292. Accuracy: 96.1500015258789
