In [1]:
import os
import random
import shutil
import time
import warnings

import torch
import torch.nn as nn
import torch.backends.cudnn as cudnn
import torch.optim

import torch.utils.data
import torchvision
import torchvision.transforms as transforms
import torchvision.datasets as datasets
import torchvision.models as models

In [2]:
import torch.distributed as dist
from torch.cuda.amp import GradScaler
from torch.cuda.amp import autocast

In [3]:
from torch.utils.tensorboard import SummaryWriter
writer = SummaryWriter()

In [4]:
!pip install wandb
import wandb
wandb.login()

Looking in indexes: https://pypi.org/simple, https://pip.repos.neuron.amazonaws.com
You should consider upgrading via the '/home/ubuntu/anaconda3/envs/pytorch_latest_p37/bin/python -m pip install --upgrade pip' command.[0m


[34m[1mwandb[0m: Currently logged in as: [33mwwblodge1[0m (use `wandb login --relogin` to force relogin)


True

In [5]:
wandb.init(project="hw9_instance")

In [6]:
# Assume that this notebook only sees one GPU.
GPU=0

In [7]:
SEED=1

In [8]:
random.seed(SEED)
torch.manual_seed(SEED)
cudnn.deterministic = True

In [9]:
torch.cuda.device_count()

1

In [10]:
START_EPOCH = 0

In [11]:
ARCH = 'resnet18'
EPOCHS = 1
LR = 0.1
MOMENTUM = 0.9
WEIGHT_DECAY = 1e-4
PRINT_FREQ = 10
TRAIN_BATCH=500
VAL_BATCH=500
WORKERS=2
TRAINDIR="data/train"
VALDIR="data/val"

global_step = 0

In [12]:
TRAINDIR="data/train"
VALDIR="data/val"

In [13]:
wandb.init(config={"epochs": EPOCHS, "batch_size": TRAIN_BATCH, "momentum": MOMENTUM, "WEIGHT_DECAY": WEIGHT_DECAY, "arch": ARCH})

VBox(children=(Label(value=' 0.00MB of 0.00MB uploaded (0.00MB deduped)\r'), FloatProgress(value=1.0, max=1.0)…

In [14]:
if not torch.cuda.is_available():
    print('GPU not detected.. did you pass through your GPU?')

In [15]:
# # how many processes in cluster?
# WORLD_SIZE = 2
# BACKEND = 'nccl'

# URL = 'tcp://172.31.24.47:1234'

In [16]:
#what is my rank?
# RANK = 0

In [17]:
# dist.init_process_group(backend = BACKEND, init_method= URL,
#                                 world_size= WORLD_SIZE, rank=RANK)

In [18]:
#torch.cuda.set_device('cpu')
torch.cuda.set_device(GPU)

In [19]:
cudnn.benchmark = True

In [20]:
# def train(train_loader, model, criterion, optimizer, epoch):
#     batch_time = AverageMeter('Time', ':6.3f')
#     data_time = AverageMeter('Data', ':6.3f')
#     losses = AverageMeter('Loss', ':.4e')
#     top1 = AverageMeter('Acc@1', ':6.2f')
#     top5 = AverageMeter('Acc@5', ':6.2f')
#     progress = ProgressMeter(
#         len(train_loader),
#         [batch_time, data_time, losses, top1, top5],
#         prefix="Epoch: [{}]".format(epoch))

#     # switch to train mode
#     model.train()

#     end = time.time()
#     for i, (images, target) in enumerate(train_loader):
#         # measure data loading time
#         data_time.update(time.time() - end)

#         if GPU is not None:
#             images = images.cuda(GPU, non_blocking=True)
#         if torch.cuda.is_available():
#             target = target.cuda(GPU, non_blocking=True)

#         # compute output
#         output = model(images)
#         loss = criterion(output, target)

#         # measure accuracy and record loss
#         acc1, acc5 = accuracy(output, target, topk=(1, 5))
#         losses.update(loss.item(), images.size(0))
#         top1.update(acc1[0], images.size(0))
#         top5.update(acc5[0], images.size(0))

#         # compute gradient and do SGD step
#         optimizer.zero_grad()
#         loss.backward()
#         optimizer.step()

#         # measure elapsed time
#         batch_time.update(time.time() - end)
#         end = time.time()

#         if i % PRINT_FREQ == 0:
#             progress.display(i)

def train(train_loader, model, criterion, optimizer, epoch):
    global global_step    
    batch_time = AverageMeter('Time', ':6.3f')
    data_time = AverageMeter('Data', ':6.3f')
    losses = AverageMeter('Loss', ':.4e')
    top1 = AverageMeter('Acc@1', ':6.2f')
    top5 = AverageMeter('Acc@5', ':6.2f')
    progress = ProgressMeter(
        len(train_loader),
        [batch_time, data_time, losses, top1, top5],
        prefix="Epoch: [{}]".format(epoch))

    # Grad Scaler
    scaler = GradScaler()
    # switch to train mode
    model.train()

    end = time.time()
    for i, (images, target) in enumerate(train_loader):
        # measure data loading time
        data_time.update(time.time() - end)
        optimizer.zero_grad()

        if GPU is not None:
            images = images.cuda(GPU, non_blocking=True)
        if torch.cuda.is_available():
            target = target.cuda(GPU, non_blocking=True)

        # compute output
        with autocast():
          output = model(images)
          loss = criterion(output, target)

        # measure accuracy and record loss
        acc1, acc5 = accuracy(output, target, topk=(1, 5))
        losses.update(loss.item(), images.size(0))
        top1.update(acc1[0], images.size(0))
        top5.update(acc5[0], images.size(0))

        # compute gradient and do SGD step
        # optimizer.zero_grad()
        # loss.backward()
        # optimizer.step()
        
        # use the scaler
        scaler.scale(loss).backward()
        scaler.step(optimizer)
        scaler.update()

        # measure elapsed time
        batch_time.update(time.time() - end)
        end = time.time()
        
        writer.add_scalar("Loss/train", loss, global_step = global_step)
        writer.add_scalar("acc1/train", top1.avg, global_step = global_step)
        writer.add_scalar("acc5/train", top5.avg, global_step = global_step)
        
        wandb.log({"Loss/train": loss, 'acc1/train': top1.avg, 'acc5/train': top5.avg})
        
        global_step = global_step + 1

        if i % PRINT_FREQ == 0:
            progress.display(i)

In [21]:
def validate(val_loader, model, criterion):
    batch_time = AverageMeter('Time', ':6.3f')
    losses = AverageMeter('Loss', ':.4e')
    top1 = AverageMeter('Acc@1', ':6.2f')
    top5 = AverageMeter('Acc@5', ':6.2f')
    progress = ProgressMeter(
        len(val_loader),
        [batch_time, losses, top1, top5],
        prefix='Test: ')

    # switch to evaluate mode
    model.eval()

    with torch.no_grad():
        end = time.time()
        for i, (images, target) in enumerate(val_loader):
            if GPU is not None:
                images = images.cuda(GPU, non_blocking=True)
            if torch.cuda.is_available():
                target = target.cuda(GPU, non_blocking=True)

            # compute output
            output = model(images)
            loss = criterion(output, target)

            # measure accuracy and record loss
            acc1, acc5 = accuracy(output, target, topk=(1, 5))
            losses.update(loss.item(), images.size(0))
            top1.update(acc1[0], images.size(0))
            top5.update(acc5[0], images.size(0))

            # measure elapsed time
            batch_time.update(time.time() - end)
            end = time.time()

            if i % PRINT_FREQ == 0:
                progress.display(i)

        # TODO: this should also be done with the ProgressMeter
        print(' * Acc@1 {top1.avg:.3f} Acc@5 {top5.avg:.3f}'
              .format(top1=top1, top5=top5))

    return top1.avg

In [22]:
def save_checkpoint(state, is_best, filename='checkpoint.pth.tar'):
    torch.save(state, filename)
    if is_best:
        shutil.copyfile(filename, 'model_best.pth.tar')

In [23]:
class AverageMeter(object):
    """Computes and stores the average and current value"""
    def __init__(self, name, fmt=':f'):
        self.name = name
        self.fmt = fmt
        self.reset()

    def reset(self):
        self.val = 0
        self.avg = 0
        self.sum = 0
        self.count = 0

    def update(self, val, n=1):
        self.val = val
        self.sum += val * n
        self.count += n
        self.avg = self.sum / self.count

    def __str__(self):
        fmtstr = '{name} {val' + self.fmt + '} ({avg' + self.fmt + '})'
        return fmtstr.format(**self.__dict__)

In [24]:
class ProgressMeter(object):
    def __init__(self, num_batches, meters, prefix=""):
        self.batch_fmtstr = self._get_batch_fmtstr(num_batches)
        self.meters = meters
        self.prefix = prefix

    def display(self, batch):
        entries = [self.prefix + self.batch_fmtstr.format(batch)]
        entries += [str(meter) for meter in self.meters]
        print('\t'.join(entries))

    def _get_batch_fmtstr(self, num_batches):
        num_digits = len(str(num_batches // 1))
        fmt = '{:' + str(num_digits) + 'd}'
        return '[' + fmt + '/' + fmt.format(num_batches) + ']'

In [25]:
def adjust_learning_rate(optimizer, epoch):
    """Sets the learning rate to the initial LR decayed by 10 every 30 epochs"""
    lr = LR * (0.1 ** (epoch // 30))
    for param_group in optimizer.param_groups:
        param_group['lr'] = lr

In [26]:
def accuracy(output, target, topk=(1,)):
    """Computes the accuracy over the k top predictions for the specified values of k"""
    with torch.no_grad():
        maxk = max(topk)
        batch_size = target.size(0)

        _, pred = output.topk(maxk, 1, True, True)
        pred = pred.t()
        correct = pred.eq(target.view(1, -1).expand_as(pred))

        res = []
        for k in topk:
            correct_k = correct[:k].reshape(-1).float().sum(0, keepdim=True)
            res.append(correct_k.mul_(100.0 / batch_size))
        return res

In [27]:
imagenet_mean_RGB = [0.47889522, 0.47227842, 0.43047404]
imagenet_std_RGB = [0.229, 0.224, 0.225]
cinic_mean_RGB = [0.47889522, 0.47227842, 0.43047404]
cinic_std_RGB = [0.24205776, 0.23828046, 0.25874835]
cifar_mean_RGB = [0.4914, 0.4822, 0.4465]
cifar_std_RGB = [0.2023, 0.1994, 0.2010]

In [28]:
normalize = transforms.Normalize(mean=cifar_mean_RGB, std=cifar_std_RGB)

In [29]:
# IMG_SIZE = 32
IMG_SIZE = 224

In [30]:
NUM_CLASSES = 1000

In [31]:
model = models.__dict__[ARCH]()

In [32]:
inf = model.fc.in_features

In [33]:
model.fc = nn.Linear(inf, NUM_CLASSES)

In [34]:
model.cuda(GPU)
# model = torch.nn.parallel.DistributedDataParallel(model, device_ids=[GPU])
# model = torch.nn.parallel.DistributedDataParallel(model)

ResNet(
  (conv1): Conv2d(3, 64, kernel_size=(7, 7), stride=(2, 2), padding=(3, 3), bias=False)
  (bn1): BatchNorm2d(64, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True)
  (relu): ReLU(inplace=True)
  (maxpool): MaxPool2d(kernel_size=3, stride=2, padding=1, dilation=1, ceil_mode=False)
  (layer1): Sequential(
    (0): BasicBlock(
      (conv1): Conv2d(64, 64, kernel_size=(3, 3), stride=(1, 1), padding=(1, 1), bias=False)
      (bn1): BatchNorm2d(64, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True)
      (relu): ReLU(inplace=True)
      (conv2): Conv2d(64, 64, kernel_size=(3, 3), stride=(1, 1), padding=(1, 1), bias=False)
      (bn2): BatchNorm2d(64, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True)
    )
    (1): BasicBlock(
      (conv1): Conv2d(64, 64, kernel_size=(3, 3), stride=(1, 1), padding=(1, 1), bias=False)
      (bn1): BatchNorm2d(64, eps=1e-05, momentum=0.1, affine=True, track_running_stats=True)
      (relu): ReLU(inplace=True)
  

In [35]:
criterion = nn.CrossEntropyLoss().cuda(GPU)

In [36]:
optimizer = torch.optim.SGD(model.parameters(), LR,
                                momentum=MOMENTUM,
                                weight_decay=WEIGHT_DECAY)

In [37]:
scheduler = torch.optim.lr_scheduler.CosineAnnealingLR(optimizer, T_max=EPOCHS)

In [38]:
transform_train = transforms.Compose([
    transforms.Resize((256,256)),
    transforms.CenterCrop(IMG_SIZE),
    transforms.RandomHorizontalFlip(),
    transforms.ToTensor(),
    transforms.Normalize(imagenet_mean_RGB, imagenet_std_RGB),
])

In [39]:
train_dataset = datasets.ImageFolder(
    TRAINDIR, transform=transform_train)

In [40]:
transform_val = transforms.Compose([
    transforms.Resize((256,256)),
    transforms.ToTensor(),
    transforms.Normalize(imagenet_mean_RGB, imagenet_std_RGB),
])

In [41]:
val_dataset = datasets.ImageFolder(
    VALDIR, transform=transform_val)

In [42]:
train_loader = torch.utils.data.DataLoader(
        train_dataset, batch_size=TRAIN_BATCH, shuffle=True,
        num_workers=WORKERS, pin_memory=True, sampler=None)

In [43]:
val_loader = torch.utils.data.DataLoader(
        val_dataset, batch_size=VAL_BATCH, shuffle=False,
        num_workers=WORKERS, pin_memory=True, sampler=None)

In [44]:
best_acc1 = 0

In [45]:
%%time
for epoch in range(START_EPOCH, 2):
#    adjust_learning_rate(optimizer, epoch)

    # train for one epoch
    train(train_loader, model, criterion, optimizer, epoch)

    # evaluate on validation set
    acc1 = validate(val_loader, model, criterion)

    # remember best acc@1 and save checkpoint
    is_best = acc1 > best_acc1
    best_acc1 = max(acc1, best_acc1)


    save_checkpoint({
        'epoch': epoch + 1,
        'arch': ARCH,
        'state_dict': model.state_dict(),
        'best_acc1': best_acc1,
        'optimizer' : optimizer.state_dict(),
    }, is_best)
    
    scheduler.step()
    print('lr: ' + str(scheduler.get_last_lr()))
    
    writer.add_scalar("lr", scheduler.get_last_lr()[0], global_step = global_step)
    
    wandb.log({'lr': scheduler.get_last_lr()[0]})

Epoch: [0][   0/2563]	Time 11.629 (11.629)	Data  4.655 ( 4.655)	Loss 7.0278e+00 (7.0278e+00)	Acc@1   0.00 (  0.00)	Acc@5   1.00 (  1.00)
Epoch: [0][  10/2563]	Time  0.912 ( 2.193)	Data  0.196 ( 0.895)	Loss 7.0486e+00 (7.0069e+00)	Acc@1   0.40 (  0.15)	Acc@5   1.00 (  0.76)
Epoch: [0][  20/2563]	Time  0.933 ( 2.188)	Data  0.216 ( 1.154)	Loss 6.9316e+00 (7.0058e+00)	Acc@1   0.20 (  0.22)	Acc@5   1.80 (  0.99)
Epoch: [0][  30/2563]	Time  2.016 ( 2.260)	Data  1.295 ( 1.320)	Loss 6.8457e+00 (6.9849e+00)	Acc@1   0.40 (  0.23)	Acc@5   2.00 (  1.08)
Epoch: [0][  40/2563]	Time  2.224 ( 2.282)	Data  1.503 ( 1.390)	Loss 6.8265e+00 (6.9522e+00)	Acc@1   0.40 (  0.33)	Acc@5   2.00 (  1.34)
Epoch: [0][  50/2563]	Time  2.903 ( 2.290)	Data  2.135 ( 1.426)	Loss 6.7876e+00 (6.9188e+00)	Acc@1   0.20 (  0.35)	Acc@5   1.20 (  1.42)
Epoch: [0][  60/2563]	Time  1.941 ( 2.293)	Data  1.216 ( 1.449)	Loss 6.7110e+00 (6.8843e+00)	Acc@1   0.80 (  0.37)	Acc@5   2.80 (  1.52)
Epoch: [0][  70/2563]	Time  0.882 ( 2.292

Epoch: [0][ 600/2563]	Time  2.519 ( 2.359)	Data  1.770 ( 1.585)	Loss 5.0334e+00 (5.8362e+00)	Acc@1   7.60 (  3.57)	Acc@5  22.60 ( 11.04)
Epoch: [0][ 610/2563]	Time  2.478 ( 2.358)	Data  1.729 ( 1.584)	Loss 4.8961e+00 (5.8233e+00)	Acc@1  10.00 (  3.64)	Acc@5  25.00 ( 11.22)
Epoch: [0][ 620/2563]	Time  1.794 ( 2.358)	Data  1.042 ( 1.584)	Loss 5.0007e+00 (5.8105e+00)	Acc@1   7.60 (  3.71)	Acc@5  21.80 ( 11.40)
Epoch: [0][ 630/2563]	Time  2.434 ( 2.359)	Data  1.685 ( 1.585)	Loss 4.9847e+00 (5.7983e+00)	Acc@1   9.00 (  3.78)	Acc@5  21.80 ( 11.56)
Epoch: [0][ 640/2563]	Time  2.880 ( 2.359)	Data  2.095 ( 1.586)	Loss 4.9237e+00 (5.7860e+00)	Acc@1   6.40 (  3.84)	Acc@5  23.00 ( 11.74)
Epoch: [0][ 650/2563]	Time  2.717 ( 2.360)	Data  1.931 ( 1.587)	Loss 5.0263e+00 (5.7738e+00)	Acc@1   7.20 (  3.91)	Acc@5  22.80 ( 11.90)
Epoch: [0][ 660/2563]	Time  2.996 ( 2.362)	Data  2.207 ( 1.588)	Loss 5.0261e+00 (5.7623e+00)	Acc@1   7.20 (  3.97)	Acc@5  23.00 ( 12.06)
Epoch: [0][ 670/2563]	Time  3.031 ( 2.362

Epoch: [0][1200/2563]	Time  1.045 ( 2.389)	Data  0.302 ( 1.616)	Loss 4.1777e+00 (5.2351e+00)	Acc@1  16.20 (  7.78)	Acc@5  38.60 ( 20.36)
Epoch: [0][1210/2563]	Time  0.856 ( 2.389)	Data  0.111 ( 1.617)	Loss 4.2768e+00 (5.2267e+00)	Acc@1  18.00 (  7.86)	Acc@5  36.80 ( 20.50)
Epoch: [0][1220/2563]	Time  0.754 ( 2.390)	Data  0.002 ( 1.617)	Loss 4.2146e+00 (5.2188e+00)	Acc@1  16.20 (  7.93)	Acc@5  37.20 ( 20.63)
Epoch: [0][1230/2563]	Time  0.752 ( 2.389)	Data  0.002 ( 1.617)	Loss 4.2961e+00 (5.2110e+00)	Acc@1  16.00 (  8.00)	Acc@5  36.40 ( 20.76)
Epoch: [0][1240/2563]	Time  0.837 ( 2.390)	Data  0.094 ( 1.617)	Loss 4.1781e+00 (5.2030e+00)	Acc@1  17.40 (  8.07)	Acc@5  38.60 ( 20.90)
Epoch: [0][1250/2563]	Time  0.836 ( 2.389)	Data  0.093 ( 1.617)	Loss 4.0894e+00 (5.1950e+00)	Acc@1  17.80 (  8.14)	Acc@5  37.60 ( 21.03)
Epoch: [0][1260/2563]	Time  0.837 ( 2.389)	Data  0.094 ( 1.617)	Loss 4.2817e+00 (5.1871e+00)	Acc@1  16.60 (  8.21)	Acc@5  35.60 ( 21.16)
Epoch: [0][1270/2563]	Time  1.322 ( 2.390

Epoch: [0][1800/2563]	Time  1.399 ( 2.398)	Data  0.655 ( 1.627)	Loss 3.7290e+00 (4.8077e+00)	Acc@1  25.00 ( 11.92)	Acc@5  47.20 ( 27.83)
Epoch: [0][1810/2563]	Time  1.786 ( 2.398)	Data  1.043 ( 1.627)	Loss 3.8253e+00 (4.8018e+00)	Acc@1  23.20 ( 11.99)	Acc@5  46.80 ( 27.95)
Epoch: [0][1820/2563]	Time  2.452 ( 2.399)	Data  1.710 ( 1.628)	Loss 3.5911e+00 (4.7956e+00)	Acc@1  27.40 ( 12.05)	Acc@5  50.40 ( 28.06)
Epoch: [0][1830/2563]	Time  3.348 ( 2.400)	Data  2.547 ( 1.629)	Loss 3.6054e+00 (4.7894e+00)	Acc@1  26.00 ( 12.12)	Acc@5  47.00 ( 28.17)
Epoch: [0][1840/2563]	Time  3.481 ( 2.400)	Data  2.684 ( 1.629)	Loss 3.6819e+00 (4.7831e+00)	Acc@1  24.40 ( 12.18)	Acc@5  49.80 ( 28.28)
Epoch: [0][1850/2563]	Time  4.019 ( 2.401)	Data  3.209 ( 1.629)	Loss 3.5274e+00 (4.7769e+00)	Acc@1  24.20 ( 12.25)	Acc@5  52.00 ( 28.39)
Epoch: [0][1860/2563]	Time  3.977 ( 2.401)	Data  3.171 ( 1.630)	Loss 3.4878e+00 (4.7708e+00)	Acc@1  27.60 ( 12.32)	Acc@5  51.20 ( 28.49)
Epoch: [0][1870/2563]	Time  4.185 ( 2.402

Epoch: [0][2400/2563]	Time  4.264 ( 2.413)	Data  3.486 ( 1.642)	Loss 3.3962e+00 (4.4756e+00)	Acc@1  29.40 ( 15.70)	Acc@5  54.80 ( 33.83)
Epoch: [0][2410/2563]	Time  4.090 ( 2.414)	Data  3.312 ( 1.643)	Loss 3.4958e+00 (4.4708e+00)	Acc@1  25.60 ( 15.76)	Acc@5  50.60 ( 33.92)
Epoch: [0][2420/2563]	Time  4.206 ( 2.414)	Data  3.422 ( 1.643)	Loss 3.3548e+00 (4.4663e+00)	Acc@1  28.60 ( 15.81)	Acc@5  54.00 ( 34.00)
Epoch: [0][2430/2563]	Time  4.144 ( 2.414)	Data  3.348 ( 1.644)	Loss 3.3087e+00 (4.4613e+00)	Acc@1  28.80 ( 15.87)	Acc@5  54.20 ( 34.09)
Epoch: [0][2440/2563]	Time  4.180 ( 2.415)	Data  3.379 ( 1.644)	Loss 3.0431e+00 (4.4565e+00)	Acc@1  31.20 ( 15.92)	Acc@5  60.40 ( 34.18)
Epoch: [0][2450/2563]	Time  4.798 ( 2.415)	Data  3.998 ( 1.645)	Loss 3.3811e+00 (4.4517e+00)	Acc@1  28.80 ( 15.98)	Acc@5  54.00 ( 34.27)
Epoch: [0][2460/2563]	Time  4.027 ( 2.415)	Data  3.233 ( 1.645)	Loss 3.2669e+00 (4.4470e+00)	Acc@1  29.40 ( 16.04)	Acc@5  54.60 ( 34.35)
Epoch: [0][2470/2563]	Time  4.147 ( 2.415

Epoch: [1][ 420/2563]	Time  4.214 ( 2.400)	Data  3.432 ( 1.633)	Loss 3.1413e+00 (3.1780e+00)	Acc@1  30.60 ( 31.35)	Acc@5  57.80 ( 57.55)
Epoch: [1][ 430/2563]	Time  4.281 ( 2.402)	Data  3.475 ( 1.635)	Loss 3.1071e+00 (3.1787e+00)	Acc@1  31.80 ( 31.34)	Acc@5  59.60 ( 57.54)
Epoch: [1][ 440/2563]	Time  3.953 ( 2.404)	Data  3.172 ( 1.637)	Loss 3.0488e+00 (3.1781e+00)	Acc@1  34.40 ( 31.36)	Acc@5  60.80 ( 57.56)
Epoch: [1][ 450/2563]	Time  4.559 ( 2.408)	Data  3.770 ( 1.641)	Loss 3.2058e+00 (3.1786e+00)	Acc@1  31.00 ( 31.35)	Acc@5  59.20 ( 57.55)
Epoch: [1][ 460/2563]	Time  4.884 ( 2.411)	Data  4.090 ( 1.644)	Loss 3.0773e+00 (3.1786e+00)	Acc@1  31.80 ( 31.34)	Acc@5  58.80 ( 57.55)
Epoch: [1][ 470/2563]	Time  4.174 ( 2.412)	Data  3.379 ( 1.645)	Loss 3.1393e+00 (3.1789e+00)	Acc@1  33.00 ( 31.35)	Acc@5  59.00 ( 57.54)
Epoch: [1][ 480/2563]	Time  4.335 ( 2.414)	Data  3.558 ( 1.647)	Loss 3.1642e+00 (3.1784e+00)	Acc@1  31.20 ( 31.37)	Acc@5  60.00 ( 57.56)
Epoch: [1][ 490/2563]	Time  4.403 ( 2.415

Epoch: [1][1020/2563]	Time  3.809 ( 2.436)	Data  3.030 ( 1.668)	Loss 3.1981e+00 (3.1803e+00)	Acc@1  29.40 ( 31.36)	Acc@5  55.20 ( 57.47)
Epoch: [1][1030/2563]	Time  3.837 ( 2.436)	Data  3.032 ( 1.668)	Loss 3.0891e+00 (3.1804e+00)	Acc@1  32.20 ( 31.37)	Acc@5  59.60 ( 57.47)
Epoch: [1][1040/2563]	Time  4.145 ( 2.436)	Data  3.352 ( 1.668)	Loss 3.1708e+00 (3.1801e+00)	Acc@1  31.20 ( 31.37)	Acc@5  58.40 ( 57.48)
Epoch: [1][1050/2563]	Time  4.027 ( 2.436)	Data  3.233 ( 1.668)	Loss 3.0665e+00 (3.1803e+00)	Acc@1  34.20 ( 31.37)	Acc@5  58.40 ( 57.47)
Epoch: [1][1060/2563]	Time  4.179 ( 2.437)	Data  3.401 ( 1.668)	Loss 3.2974e+00 (3.1809e+00)	Acc@1  29.00 ( 31.36)	Acc@5  56.00 ( 57.46)
Epoch: [1][1070/2563]	Time  4.110 ( 2.437)	Data  3.323 ( 1.668)	Loss 3.2011e+00 (3.1815e+00)	Acc@1  29.00 ( 31.35)	Acc@5  55.60 ( 57.46)
Epoch: [1][1080/2563]	Time  4.098 ( 2.437)	Data  3.320 ( 1.669)	Loss 3.1302e+00 (3.1812e+00)	Acc@1  33.80 ( 31.35)	Acc@5  58.40 ( 57.47)
Epoch: [1][1090/2563]	Time  4.096 ( 2.438

Epoch: [1][1620/2563]	Time  4.092 ( 2.447)	Data  3.316 ( 1.679)	Loss 3.1030e+00 (3.1822e+00)	Acc@1  35.00 ( 31.35)	Acc@5  58.40 ( 57.42)
Epoch: [1][1630/2563]	Time  4.184 ( 2.447)	Data  3.405 ( 1.679)	Loss 3.3026e+00 (3.1822e+00)	Acc@1  30.60 ( 31.36)	Acc@5  55.00 ( 57.42)
Epoch: [1][1640/2563]	Time  3.998 ( 2.447)	Data  3.191 ( 1.678)	Loss 3.3924e+00 (3.1820e+00)	Acc@1  29.40 ( 31.36)	Acc@5  54.00 ( 57.42)
Epoch: [1][1650/2563]	Time  4.037 ( 2.447)	Data  3.232 ( 1.678)	Loss 3.1356e+00 (3.1819e+00)	Acc@1  31.60 ( 31.36)	Acc@5  60.00 ( 57.42)
Epoch: [1][1660/2563]	Time  4.157 ( 2.447)	Data  3.375 ( 1.679)	Loss 3.2553e+00 (3.1821e+00)	Acc@1  27.60 ( 31.36)	Acc@5  56.00 ( 57.42)
Epoch: [1][1670/2563]	Time  3.989 ( 2.447)	Data  3.198 ( 1.679)	Loss 3.2347e+00 (3.1819e+00)	Acc@1  32.00 ( 31.36)	Acc@5  59.80 ( 57.42)
Epoch: [1][1680/2563]	Time  4.421 ( 2.448)	Data  3.630 ( 1.679)	Loss 3.0443e+00 (3.1820e+00)	Acc@1  33.80 ( 31.36)	Acc@5  59.40 ( 57.42)
Epoch: [1][1690/2563]	Time  4.035 ( 2.447

Epoch: [1][2220/2563]	Time  4.036 ( 2.454)	Data  3.256 ( 1.686)	Loss 3.2492e+00 (3.1837e+00)	Acc@1  30.40 ( 31.34)	Acc@5  55.40 ( 57.39)
Epoch: [1][2230/2563]	Time  4.220 ( 2.454)	Data  3.420 ( 1.685)	Loss 3.2476e+00 (3.1838e+00)	Acc@1  32.00 ( 31.34)	Acc@5  56.20 ( 57.39)
Epoch: [1][2240/2563]	Time  3.861 ( 2.453)	Data  3.066 ( 1.685)	Loss 3.2239e+00 (3.1840e+00)	Acc@1  31.60 ( 31.33)	Acc@5  54.80 ( 57.39)
Epoch: [1][2250/2563]	Time  4.548 ( 2.454)	Data  3.767 ( 1.685)	Loss 3.2160e+00 (3.1840e+00)	Acc@1  31.00 ( 31.33)	Acc@5  55.20 ( 57.39)
Epoch: [1][2260/2563]	Time  4.074 ( 2.454)	Data  3.272 ( 1.685)	Loss 3.1991e+00 (3.1841e+00)	Acc@1  30.80 ( 31.34)	Acc@5  56.40 ( 57.39)
Epoch: [1][2270/2563]	Time  4.053 ( 2.453)	Data  3.272 ( 1.685)	Loss 3.1704e+00 (3.1842e+00)	Acc@1  31.40 ( 31.34)	Acc@5  55.20 ( 57.39)
Epoch: [1][2280/2563]	Time  4.206 ( 2.453)	Data  3.404 ( 1.685)	Loss 3.1758e+00 (3.1843e+00)	Acc@1  29.00 ( 31.33)	Acc@5  57.80 ( 57.39)
Epoch: [1][2290/2563]	Time  4.078 ( 2.453

In [46]:

writer.close()
%load_ext tensorboard
%tensorboard --logdir=runs

Reusing TensorBoard on port 6006 (pid 4802), started 5:21:57 ago. (Use '!kill 4802' to kill it.)