In [None]:
import torch
import torch.nn as nn
import torchvision.transforms as transforms
import torchvision.datasets as datasets

In [None]:
train_dataset = datasets.MNIST(root='./data',transform=transforms.ToTensor(),download=True)
test_dataset = datasets.MNIST(root='./data',train=False,transform=transforms.ToTensor())

In [None]:
batch_size,n_iters = 100,3000
num_epochs = int(n_iters/(len(train_dataset)/batch_size))

train_loader = torch.utils.data.DataLoader(dataset=train_dataset,batch_size=batch_size,shuffle=True)
test_loader = torch.utils.data.DataLoader(dataset=test_dataset,batch_size=batch_size,shuffle=False)

In [None]:
class FeedForwardNeuralNetwork(nn.Module):
    def __init__(self,input_dim, hidden_dim, output_dim):
        super(FeedForwardNeuralNetwork,self).__init__()

        self.fc1 = nn.Linear(input_dim,hidden_dim)

        self.sigmoid = nn.Sigmoid()
        self.fc2 = nn.Linear(hidden_dim,output_dim)

    def forward(self,x):
        out = self.fc1(x)
        out = self.sigmoid(out)
        out = self.fc2(out)
        return out


In [None]:
input_dim = 28**2
hidden_dim = 100
output_dim = 10
model = FeedForwardNeuralNetwork(input_dim=input_dim,hidden_dim=hidden_dim,output_dim=output_dim)

In [None]:
criterion = nn.CrossEntropyLoss()

In [None]:
learning_Rate = 0.1
optimizer = torch.optim.SGD(model.parameters(),lr=learning_Rate)

In [None]:
iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Iteration: 500. Loss: 0.4963933527469635. Accuracy: 86
Iteration: 1000. Loss: 0.3637200891971588. Accuracy: 89
Iteration: 1500. Loss: 0.29671013355255127. Accuracy: 90
Iteration: 2000. Loss: 0.2967323362827301. Accuracy: 91
Iteration: 2500. Loss: 0.2763761281967163. Accuracy: 91
Iteration: 3000. Loss: 0.5291683673858643. Accuracy: 92


In [None]:
train_dataset = datasets.MNIST(root='./data',transform=transforms.ToTensor(),download=True)
test_dataset = datasets.MNIST(root='./data',train=False,transform=transforms.ToTensor())

batch_size,n_iters = 100,3000
num_epochs = int(n_iters/(len(train_dataset)/batch_size))

train_loader = torch.utils.data.DataLoader(dataset=train_dataset,batch_size=batch_size,shuffle=True)
test_loader = torch.utils.data.DataLoader(dataset=test_dataset,batch_size=batch_size,shuffle=False)

class FeedForwardNeuralNetwork(nn.Module):
    def __init__(self,input_dim, hidden_dim, output_dim):
        super(FeedForwardNeuralNetwork,self).__init__()

        self.fc1 = nn.Linear(input_dim,hidden_dim)

        self.tanh = nn.Tanh()
        self.fc2 = nn.Linear(hidden_dim,output_dim)

    def forward(self,x):
        out = self.fc1(x)
        out = self.tanh(out)
        out = self.fc2(out)
        return out

input_dim = 28**2
hidden_dim = 100
output_dim = 10
model = FeedForwardNeuralNetwork(input_dim=input_dim,hidden_dim=hidden_dim,output_dim=output_dim)

criterion = nn.CrossEntropyLoss()
learning_Rate = 0.1
optimizer = torch.optim.SGD(model.parameters(),lr=learning_Rate)

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Iteration: 500. Loss: 0.5495707988739014. Accuracy: 91
Iteration: 1000. Loss: 0.2172483503818512. Accuracy: 92
Iteration: 1500. Loss: 0.16044455766677856. Accuracy: 93
Iteration: 2000. Loss: 0.2312755435705185. Accuracy: 93
Iteration: 2500. Loss: 0.21641124784946442. Accuracy: 94
Iteration: 3000. Loss: 0.1174468919634819. Accuracy: 95


In [None]:
train_dataset = datasets.MNIST(root='./data',transform=transforms.ToTensor(),download=True)
test_dataset = datasets.MNIST(root='./data',train=False,transform=transforms.ToTensor())

batch_size,n_iters = 100,3000
num_epochs = int(n_iters/(len(train_dataset)/batch_size))

train_loader = torch.utils.data.DataLoader(dataset=train_dataset,batch_size=batch_size,shuffle=True)
test_loader = torch.utils.data.DataLoader(dataset=test_dataset,batch_size=batch_size,shuffle=False)

class FeedForwardNeuralNetwork(nn.Module):
    def __init__(self,input_dim, hidden_dim, output_dim):
        super(FeedForwardNeuralNetwork,self).__init__()

        self.fc1 = nn.Linear(input_dim,hidden_dim)

        self.relu = nn.ReLU()
        self.fc2 = nn.Linear(hidden_dim,output_dim)

    def forward(self,x):
        out = self.fc1(x)
        out = self.relu(out)
        out = self.fc2(out)
        return out

input_dim = 28**2
hidden_dim = 100
output_dim = 10
model = FeedForwardNeuralNetwork(input_dim=input_dim,hidden_dim=hidden_dim,output_dim=output_dim)

criterion = nn.CrossEntropyLoss()
learning_Rate = 0.1
optimizer = torch.optim.SGD(model.parameters(),lr=learning_Rate)

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Iteration: 500. Loss: 0.41141125559806824. Accuracy: 91
Iteration: 1000. Loss: 0.19843018054962158. Accuracy: 92
Iteration: 1500. Loss: 0.20181341469287872. Accuracy: 94
Iteration: 2000. Loss: 0.17491742968559265. Accuracy: 94
Iteration: 2500. Loss: 0.1918606162071228. Accuracy: 95
Iteration: 3000. Loss: 0.24881862103939056. Accuracy: 95


In [None]:
train_dataset = datasets.MNIST(root='./data',transform=transforms.ToTensor(),download=True)
test_dataset = datasets.MNIST(root='./data',train=False,transform=transforms.ToTensor())

batch_size,n_iters = 100,3000
num_epochs = int(n_iters/(len(train_dataset)/batch_size))

train_loader = torch.utils.data.DataLoader(dataset=train_dataset,batch_size=batch_size,shuffle=True)
test_loader = torch.utils.data.DataLoader(dataset=test_dataset,batch_size=batch_size,shuffle=False)

class FeedForwardNeuralNetwork(nn.Module):
    def __init__(self,input_dim, hidden_dim, output_dim):
        super(FeedForwardNeuralNetwork,self).__init__()

        self.fc1 = nn.Linear(input_dim,hidden_dim)

        self.relu1 = nn.ReLU()
        self.fc2 = nn.Linear(hidden_dim,hidden_dim)

        self.relu2 = nn.ReLU()
        self.fc3 = nn.Linear(hidden_dim,output_dim)

    def forward(self,x):
        out = self.fc1(x)
        out = self.relu1(out)
        out = self.fc2(out)
        out = self.relu2(out)
        out = self.fc3(out)
        return out

input_dim = 28**2
hidden_dim = 100
output_dim = 10
model = FeedForwardNeuralNetwork(input_dim=input_dim,hidden_dim=hidden_dim,output_dim=output_dim)

criterion = nn.CrossEntropyLoss()
learning_Rate = 0.1
optimizer = torch.optim.SGD(model.parameters(),lr=learning_Rate)

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Iteration: 500. Loss: 0.4337555766105652. Accuracy: 91
Iteration: 1000. Loss: 0.16706930100917816. Accuracy: 93
Iteration: 1500. Loss: 0.2003202885389328. Accuracy: 94
Iteration: 2000. Loss: 0.07536744326353073. Accuracy: 95
Iteration: 2500. Loss: 0.1584356129169464. Accuracy: 96
Iteration: 3000. Loss: 0.17269065976142883. Accuracy: 96


In [None]:
train_dataset = datasets.MNIST(root='./data',transform=transforms.ToTensor(),download=True)
test_dataset = datasets.MNIST(root='./data',train=False,transform=transforms.ToTensor())

batch_size,n_iters = 100,3000
num_epochs = int(n_iters/(len(train_dataset)/batch_size))

train_loader = torch.utils.data.DataLoader(dataset=train_dataset,batch_size=batch_size,shuffle=True)
test_loader = torch.utils.data.DataLoader(dataset=test_dataset,batch_size=batch_size,shuffle=False)

class FeedForwardNeuralNetwork(nn.Module):
    def __init__(self,input_dim, hidden_dim, output_dim):
        super(FeedForwardNeuralNetwork,self).__init__()

        self.fc1 = nn.Linear(input_dim,hidden_dim)

        self.relu1 = nn.ReLU()
        self.fc2 = nn.Linear(hidden_dim,hidden_dim)

        self.relu2 = nn.ReLU()
        self.fc3 = nn.Linear(hidden_dim,hidden_dim)

        self.relu3 = nn.ReLU()
        self.fc4 = nn.Linear(hidden_dim,output_dim)

    def forward(self,x):
        out = self.fc1(x)
        out = self.relu1(out)
        out = self.fc2(out)
        out = self.relu2(out)
        out = self.fc3(out)
        out = self.relu3(out)
        out = self.fc4(out)
        return out

input_dim = 28**2
hidden_dim = 100
output_dim = 10
model = FeedForwardNeuralNetwork(input_dim=input_dim,hidden_dim=hidden_dim,output_dim=output_dim)

criterion = nn.CrossEntropyLoss()
learning_Rate = 0.1
optimizer = torch.optim.SGD(model.parameters(),lr=learning_Rate)

iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_()

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_()

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Iteration: 500. Loss: 0.33805596828460693. Accuracy: 91
Iteration: 1000. Loss: 0.3607591688632965. Accuracy: 93
Iteration: 1500. Loss: 0.13835036754608154. Accuracy: 95
Iteration: 2000. Loss: 0.053997211158275604. Accuracy: 96
Iteration: 2500. Loss: 0.09074047952890396. Accuracy: 96
Iteration: 3000. Loss: 0.0835341066122055. Accuracy: 96


In [None]:
train_dataset = datasets.MNIST(root='./data',transform=transforms.ToTensor(),download=True)
test_dataset = datasets.MNIST(root='./data',train=False,transform=transforms.ToTensor())

batch_size,n_iters = 100,3000
num_epochs = int(n_iters/(len(train_dataset)/batch_size))

train_loader = torch.utils.data.DataLoader(dataset=train_dataset,batch_size=batch_size,shuffle=True)
test_loader = torch.utils.data.DataLoader(dataset=test_dataset,batch_size=batch_size,shuffle=False)

class FeedForwardNeuralNetwork(nn.Module):
    def __init__(self,input_dim, hidden_dim, output_dim):
        super(FeedForwardNeuralNetwork,self).__init__()

        self.fc1 = nn.Linear(input_dim,hidden_dim)

        self.relu1 = nn.ReLU()
        self.fc2 = nn.Linear(hidden_dim,hidden_dim)

        self.relu2 = nn.ReLU()
        self.fc3 = nn.Linear(hidden_dim,hidden_dim)

        self.relu3 = nn.ReLU()
        self.fc4 = nn.Linear(hidden_dim,output_dim)

    def forward(self,x):
        out = self.fc1(x)
        out = self.relu1(out)
        out = self.fc2(out)
        out = self.relu2(out)
        out = self.fc3(out)
        out = self.relu3(out)
        out = self.fc4(out)
        return out

input_dim = 28**2
hidden_dim = 100
output_dim = 10
model = FeedForwardNeuralNetwork(input_dim=input_dim,hidden_dim=hidden_dim,output_dim=output_dim)

device = torch.device("cuda:0" if torch.cuda.is_available() else "cpu")
model.to(device)

criterion = nn.CrossEntropyLoss()
learning_Rate = 0.1
optimizer = torch.optim.SGD(model.parameters(),lr=learning_Rate)


iter = 0
for epoch in range(num_epochs):
    for i, (images, labels) in enumerate(train_loader):
        # Load images with gradient accumulation capabilities
        images = images.view(-1, 28*28).requires_grad_().to(device)
        labels = labels.to(device)

        # Clear gradients w.r.t. parameters
        optimizer.zero_grad()

        # Forward pass to get output/logits
        outputs = model(images)

        # Calculate Loss: softmax --> cross entropy loss
        loss = criterion(outputs, labels)

        # Getting gradients w.r.t. parameters
        loss.backward()

        # Updating parameters
        optimizer.step()

        iter += 1

        if iter % 500 == 0:
            # Calculate Accuracy         
            correct = 0
            total = 0
            # Iterate through test dataset
            for images, labels in test_loader:
                # Load images with gradient accumulation capabilities
                images = images.view(-1, 28*28).requires_grad_().to(device)

                # Forward pass only to get logits/output
                outputs = model(images)

                # Get predictions from the maximum value
                _, predicted = torch.max(outputs.data, 1)

                # Total number of labels
                total += labels.size(0)

                # Total correct predictions
                if torch.cuda.is_available():
                    correct += (predicted.cpu() == labels.cpu()).sum()
                else:
                    correct += (predicted == labels).sum()

            accuracy = 100 * correct / total

            # Print Loss
            print('Iteration: {}. Loss: {}. Accuracy: {}'.format(iter, loss.item(), accuracy))

Iteration: 500. Loss: 0.3662842810153961. Accuracy: 90
Iteration: 1000. Loss: 0.1286233365535736. Accuracy: 93
Iteration: 1500. Loss: 0.2579042911529541. Accuracy: 95
Iteration: 2000. Loss: 0.08410173654556274. Accuracy: 96
Iteration: 2500. Loss: 0.0626690611243248. Accuracy: 96
Iteration: 3000. Loss: 0.11885710805654526. Accuracy: 96
