In [1]:
import torch
import torchvision
import torchvision.transforms as transforms
import numpy as np
from torch.utils.data.sampler import SubsetRandomSampler

train_on_gpu = torch.cuda.is_available()

if not train_on_gpu:
    print('Cuda is not available. Training on CPU...')
else:
    print('Cuda is available! Training on GPU...')

Cuda is available! Training on GPU...


In [2]:
num_workers = 0
batch_size = 64
valid_size = 0.2

transform = transforms.ToTensor()

train_data = torchvision.datasets.EMNIST(
    root='data', split='byclass', train=True, download=True, transform=transform
)
test_data = torchvision.datasets.EMNIST(
    root='data', split='byclass', train=False, download=True, transform=transform
)

num_train = len(train_data)
indices = list(range(num_train))
np.random.shuffle(indices)
split = int(np.floor(valid_size * num_train))
train_idx, valid_idx = indices[split:], indices[:split]

train_sampler = SubsetRandomSampler(train_idx)
valid_sampler = SubsetRandomSampler(valid_idx)

train_loader = torch.utils.data.DataLoader(
    train_data, batch_size=batch_size, sampler=train_sampler, num_workers=num_workers
)
valid_loader = torch.utils.data.DataLoader(
    train_data, batch_size=batch_size, sampler=valid_sampler, num_workers=num_workers
)
test_loader = torch.utils.data.DataLoader(
    test_data, batch_size=batch_size, num_workers=num_workers
)

classes = train_data.classes

In [None]:
import torch.nn as nn
import torch.nn.functional as F

class Net(nn.Module):
    def __init__(self):
        super(Net, self).__init__()
        # Conv Block 1: 1x28x28 -> 32x24x24 -> 32x12x12
        self.conv1 = nn.Conv2d(1, 32, kernel_size=5)
        self.pool1 = nn.MaxPool2d(2)
        
        # Conv Block 2: 32x12x12 -> 64x8x8 -> 64x4x4
        self.conv2 = nn.Conv2d(32, 64, kernel_size=5)
        self.pool2 = nn.MaxPool2d(2)
        
        # FC Layers
        self.fc1 = nn.Linear(64 * 4 * 4, 512) # 64 channels * 4x4 feature map
        self.dropout = nn.Dropout(0.5)
        self.fc2 = nn.Linear(512, 10) # 10 output classes for digits

    def forward(self, x):
        x = self.pool1(F.relu(self.conv1(x)))
        x = self.pool2(F.relu(self.conv2(x)))
        x = x.view(-1, 64 * 4 * 4) # Flatten the tensor
        x = F.relu(self.fc1(x))
        x = self.dropout(x)
        x = self.fc2(x)
        return F.log_softmax(x, dim=1)

model = Net()
print("\n--- Model Architecture ---")
print(model)

if train_on_gpu:
  model.cuda()


--- Model Architecture ---
Net(
  (conv1): Conv2d(1, 32, kernel_size=(5, 5), stride=(1, 1))
  (pool1): MaxPool2d(kernel_size=2, stride=2, padding=0, dilation=1, ceil_mode=False)
  (conv2): Conv2d(32, 64, kernel_size=(5, 5), stride=(1, 1))
  (pool2): MaxPool2d(kernel_size=2, stride=2, padding=0, dilation=1, ceil_mode=False)
  (fc1): Linear(in_features=1024, out_features=512, bias=True)
  (dropout): Dropout(p=0.5, inplace=False)
  (fc2): Linear(in_features=512, out_features=10, bias=True)
)


In [28]:
import torch.optim as optim

criterion = nn.CrossEntropyLoss()

optimizer = optim.Adam(model.parameters(), lr=0.0003)

In [29]:
# number of epochs to train the model, you decide the number
n_epochs = 15

valid_loss_min = np.inf # track change in validation loss

for epoch in range(1, n_epochs+1):

    # keep track of training and validation loss
    train_loss = 0.0
    valid_loss = 0.0
    
    ###################
    # train the model #
    ###################
    model.train()
    for batch_idx, (data, target) in enumerate(train_loader):
        # move tensors to GPU if CUDA is available
        if train_on_gpu:
            data, target = data.cuda(), target.cuda()
        # clear the gradients of all optimized variables
        optimizer.zero_grad()
        # forward pass: compute predicted outputs by passing inputs to the model
        output = model(data)
        # calculate the batch loss
        loss = criterion(output, target)
        # backward pass: compute gradient of the loss with respect to model parameters
        loss.backward()
        # perform a single optimization step (parameter update)
        optimizer.step()
        # update training loss
        train_loss += loss.item()*data.size(0)
        
    ######################    
    # validate the model #
    ######################
    model.eval()
    for batch_idx, (data, target) in enumerate(valid_loader):
        # move tensors to GPU if CUDA is available
        if train_on_gpu:
            data, target = data.cuda(), target.cuda()
        # forward pass: compute predicted outputs by passing inputs to the model
        output = model(data)
        # calculate the batch loss
        loss = criterion(output, target)
        # update average validation loss 
        valid_loss += loss.item()*data.size(0)
    
    # calculate average losses
    train_loss = train_loss/len(train_loader.sampler)
    valid_loss = valid_loss/len(valid_loader.sampler)
        
    # print training/validation statistics 
    print('Epoch: {} \tTraining Loss: {:.6f} \tValidation Loss: {:.6f}'.format(
        epoch, train_loss, valid_loss))
    
    # save model if validation loss has decreased
    if valid_loss <= valid_loss_min:
        print('Validation loss decreased ({:.6f} --> {:.6f}).  Saving model ...'.format(
        valid_loss_min,
        valid_loss))
        torch.save(model.state_dict(), 'model_trained.pt')
        valid_loss_min = valid_loss
    else:
        print('Validation loss has not decreased. Ending training')
        break

RuntimeError: GET was unable to find an engine to execute this computation

In [22]:
model.load_state_dict(torch.load('./model_trained.pt'))

<All keys matched successfully>

In [23]:
# track test loss
test_loss = 0.0
class_correct = list(0. for i in range(64))
class_total = list(0. for i in range(64))

model.eval()
# iterate over test data
for batch_idx, (data, target) in enumerate(test_loader):
    # move tensors to GPU if CUDA is available
    if train_on_gpu:
        data, target = data.cuda(), target.cuda()
    # forward pass: compute predicted outputs by passing inputs to the model
    output = model(data)
    # calculate the batch loss
    loss = criterion(output, target)
    # update test loss 
    test_loss += loss.item()*data.size(0)
    # convert output probabilities to predicted class
    _, pred = torch.max(output, 1)    
    # compare predictions to true label
    correct_tensor = pred.eq(target.data.view_as(pred))
    correct = np.squeeze(correct_tensor.numpy()) if not train_on_gpu else np.squeeze(correct_tensor.cpu().numpy())
    # calculate test accuracy for each object class
    for i in range(len(target)):
        label = target.data[i]
        class_correct[label] += correct[i].item()
        class_total[label] += 1

# average test loss
test_loss = test_loss/len(test_loader.dataset)
print('Test Loss: {:.6f}\n'.format(test_loss))

for i in range(len(classes)):
    if class_total[i] > 0:
        print('Test Accuracy of %5s: %2d%% (%2d/%2d)' % (
            classes[i], 100 * class_correct[i] / class_total[i],
            np.sum(class_correct[i]), np.sum(class_total[i])))
    else:
        print('Test Accuracy of %5s: N/A (no training examples)' % (classes[i]))

print('\nTest Accuracy (Overall): %2d%% (%2d/%2d)' % (
    100. * np.sum(class_correct) / np.sum(class_total),
    np.sum(class_correct), np.sum(class_total)))

Test Loss: 0.317001

Test Accuracy of     0: 84% (4866/5778)
Test Accuracy of     1: 86% (5498/6330)
Test Accuracy of     2: 98% (5801/5869)
Test Accuracy of     3: 99% (5945/5969)
Test Accuracy of     4: 98% (5526/5619)
Test Accuracy of     5: 94% (4883/5190)
Test Accuracy of     6: 98% (5647/5705)
Test Accuracy of     7: 99% (6105/6139)
Test Accuracy of     8: 99% (5590/5633)
Test Accuracy of     9: 98% (5614/5686)
Test Accuracy of     A: 99% (1056/1062)
Test Accuracy of     B: 97% (633/648)
Test Accuracy of     C: 97% (1696/1739)
Test Accuracy of     D: 92% (718/779)
Test Accuracy of     E: 99% (845/851)
Test Accuracy of     F: 96% (1396/1440)
Test Accuracy of     G: 91% (410/447)
Test Accuracy of     H: 97% (507/521)
Test Accuracy of     I: 59% (1226/2048)
Test Accuracy of     J: 86% (541/626)
Test Accuracy of     K: 76% (292/382)
Test Accuracy of     L: 94% (768/810)
Test Accuracy of     M: 99% (1476/1485)
Test Accuracy of     N: 99% (1345/1351)
Test Accuracy of     O: 54% (2284/4

In [25]:
import pygame
import sys
from PIL import Image
import matplotlib.pyplot as plt
import torch

CHARACTER_MAP = [
    '0', '1', '2', '3', '4', '5', '6', '7', '8', '9',
    'A', 'B', 'C', 'D', 'E', 'F', 'G', 'H', 'I', 'J', 'K', 'L', 'M',
    'N', 'O', 'P', 'Q', 'R', 'S', 'T', 'U', 'V', 'W', 'X', 'Y', 'Z',
    'a', 'b', 'c', 'd', 'e', 'f', 'g', 'h', 'i', 'j', 'k', 'l', 'm',
    'n', 'o', 'p', 'q', 'r', 's', 't', 'u', 'v', 'w', 'x', 'y', 'z'
]

pygame.init()

WIDTH, HEIGHT = 250, 250
LINE_WIDTH = 15
WHITE = (255, 255, 255)
BLACK = (0, 0, 0)

screen = pygame.display.set_mode((WIDTH, HEIGHT))
pygame.display.set_caption("Draw a Character (ENTER to predict, C to clear)")

drawing = False
running = True
screen.fill(BLACK)

while running:
  for event in pygame.event.get():
    if event.type == pygame.QUIT:
      running = False

    elif event.type == pygame.MOUSEBUTTONDOWN:
      drawing = True
      pygame.draw.circle(screen, WHITE, event.pos, LINE_WIDTH // 2)

    elif event.type == pygame.MOUSEBUTTONUP:
      drawing = False

    elif event.type == pygame.MOUSEMOTION:
      if drawing:
        pygame.draw.circle(screen, WHITE, event.pos, LINE_WIDTH // 2)

    elif event.type == pygame.KEYDOWN:
      if event.key == pygame.K_c:
        screen.fill(BLACK)
        pygame.display.set_caption("Draw a Character (ENTER to predict, C to clear)")
      if event.key == pygame.K_q:
        running = False

      if event.key == pygame.K_RETURN:
        try:
          pixel_data = pygame.surfarray.array3d(screen)
          pil_image = Image.fromarray(pixel_data.transpose(1, 0, 2)).convert('L')
          resized_image = pil_image.resize((28, 28))

          rotated_image = resized_image.rotate(-90, resample=Image.BICUBIC)
          flipped_image = rotated_image.transpose(Image.FLIP_LEFT_RIGHT)

          image_tensor = transform(flipped_image).unsqueeze(0)
          model.eval()
          if train_on_gpu:
            image_tensor = image_tensor.cuda()

          with torch.no_grad():
            output = model(image_tensor)
            _, pred = torch.max(output, 1)
            prediction_index = pred.item()

            if 0 <= prediction_index < len(CHARACTER_MAP):
                prediction_char = CHARACTER_MAP[prediction_index]
            else:
                prediction_char = '?'

          print(f'--- Prediction: {prediction_char} (index: {prediction_index}) ---')
          pygame.display.set_caption(f"Prediction: {prediction_char} (Press C to clear)")

        except Exception as e:
          print(f"An error occurred during prediction: {e}")

  pygame.display.flip()

pygame.quit()

--- Prediction: A (index: 10) ---
--- Prediction: O (index: 24) ---
--- Prediction: O (index: 24) ---
--- Prediction: 0 (index: 0) ---
--- Prediction: I (index: 18) ---
--- Prediction: 1 (index: 1) ---
--- Prediction: 2 (index: 2) ---
--- Prediction: 3 (index: 3) ---
--- Prediction: I (index: 18) ---
--- Prediction: 4 (index: 4) ---
--- Prediction: 4 (index: 4) ---
--- Prediction: 5 (index: 5) ---
--- Prediction: 6 (index: 6) ---
--- Prediction: 7 (index: 7) ---
--- Prediction: 8 (index: 8) ---
--- Prediction: g (index: 42) ---
--- Prediction: 9 (index: 9) ---
--- Prediction: 9 (index: 9) ---
--- Prediction: 9 (index: 9) ---
--- Prediction: a (index: 36) ---
--- Prediction: C (index: 12) ---
--- Prediction: a (index: 36) ---
--- Prediction: b (index: 37) ---
--- Prediction: E (index: 14) ---
--- Prediction: E (index: 14) ---
--- Prediction: C (index: 12) ---
--- Prediction: d (index: 39) ---
--- Prediction: C (index: 12) ---
--- Prediction: e (index: 40) ---
--- Prediction: I (index: 1