In [1]:
import sys
if not sys.warnoptions:
    import warnings
    warnings.simplefilter("ignore")

import torch
import torch.optim as optim
from torchvision import transforms
from torch.utils.data import DataLoader
from tqdm import tqdm

from data_utility import *
from data_utils import *
from loss import *
from train import *
from deeplab_model.deeplab import *
from sync_batchnorm import convert_model
import datetime

%matplotlib inline
%load_ext autoreload
%autoreload 2

In [2]:
USE_GPU = True
NUM_WORKERS = 12
BATCH_SIZE = 2 

dtype = torch.float32 
# define dtype, float is space efficient than double

if USE_GPU and torch.cuda.is_available():
    
    device = torch.device('cuda')
    
    torch.backends.cudnn.benchmark = True
    torch.backends.cudnn.enabled = True
    # magic flag that accelerate
    
    print('using GPU for training')
else:
    device = torch.device('cpu')
    print('using CPU for training')

using GPU for training


In [3]:
train_dataset = pyramid_dataset(data_type = 'nii_train', 
                transform=transforms.Compose([
                random_affine(90, 15),
                random_filp(0.5)]))
# do data augumentation on train dataset

validation_dataset = pyramid_dataset(data_type = 'nii_test', 
                transform=None)
# no data augumentation on validation dataset

train_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True,
                    num_workers=NUM_WORKERS)
validation_loader = DataLoader(validation_dataset, batch_size=BATCH_SIZE, shuffle=True,
                    num_workers=NUM_WORKERS) # drop_last
# loaders come with auto batch division and multi-thread acceleration

In [4]:
'''
deeplab = DeepLab_ELU(output_stride=8)
deeplab = nn.DataParallel(deeplab)
deeplab = convert_model(deeplab)
deeplab = deeplab.to(device=device, dtype=dtype)
#shape_test(icnet1, True)
# create the model, by default model type is float, use model.double(), model.float() to convert
# move the model to desirable device

optimizer = optim.Adam(deeplab.parameters(), lr=1e-2)
scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=10)
epoch = 0

# create an optimizer object
# note that only the model_2 params and model_4 params will be optimized by optimizer
'''

"\ndeeplab = DeepLab_ELU(output_stride=8)\ndeeplab = nn.DataParallel(deeplab)\ndeeplab = convert_model(deeplab)\ndeeplab = deeplab.to(device=device, dtype=dtype)\n#shape_test(icnet1, True)\n# create the model, by default model type is float, use model.double(), model.float() to convert\n# move the model to desirable device\n\noptimizer = optim.Adam(deeplab.parameters(), lr=1e-2)\nscheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=10)\nepoch = 0\n\n# create an optimizer object\n# note that only the model_2 params and model_4 params will be optimized by optimizer\n"

In [None]:

deeplab = DeepLab_ELU(output_stride=8)
deeplab = nn.DataParallel(deeplab)
deeplab = convert_model(deeplab)

#checkpoint = torch.load('../deeplab_save/2019-07-29 04:00:14.630172.pth') # second best
#checkpoint = torch.load('../deeplab_save/2019-07-28 23:47:36.279119.pth') # second best
#checkpoint = torch.load('../deeplab_save/2019-07-29 00:15:49.271222.pth') # best
#checkpoint = torch.load('../deeplab_save/2019-07-29 00:44:11.825872.pth')
checkpoint = torch.load('../deeplab_output_8_elu_save/2019-08-20 21:21:15.471972 epoch: 350.pth') # latest one

deeplab.load_state_dict(checkpoint['state_dict_1'])
deeplab = deeplab.to(device=device, dtype=dtype)

optimizer = optim.Adam(deeplab.parameters(), lr=1e-2)
optimizer.load_state_dict(checkpoint['optimizer'])

scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.1, patience=25)
scheduler.load_state_dict(checkpoint['scheduler'])

epoch = checkpoint['epoch']
print(epoch)
for param_group in optimizer.param_groups:
    print(param_group['lr'])


330
0.01


In [None]:
epochs = 5000

min_val = .0621

record = open('train_deeplab_output_8_elu.txt','a+')

logger = {'train':[], 'validation_1': []}

for e in tqdm(range(epoch + 1, epochs)):
# iter over epoches

    epoch_loss = 0
        
    for t, batch in enumerate(train_loader):
    # iter over the train mini batches
    
        deeplab.train()
        # Set the model flag to train
        # 1. enable dropout
        # 2. batchnorm behave differently in train and test
        
        image_1 = batch['image1_data'].to(device=device, dtype=dtype)
        label_1 = batch['image1_label'].to(device=device, dtype=dtype)
        # move data to device, convert dtype to desirable dtype
        
        out_1 = deeplab(image_1)
        # do the inference

        loss_1 = dice_loss_3(out_1, label_1)
        # calculate loss
        
        epoch_loss += loss_1.item()
        # record minibatch loss to epoch loss
        
        optimizer.zero_grad()
        # set the model parameter gradient to zero
        
        loss_1.backward()
        # calculate the gradient wrt loss
        optimizer.step()
        #scheduler.step(loss_1)
        # take a gradient descent step
        
    outstr = 'Epoch {0} finished ! Training Loss: {1:.4f}'.format(e, epoch_loss/(t+1)) + '\n'
    
    logger['train'].append(epoch_loss/(t+1))
    
    print(outstr)
    record.write(outstr)
    record.flush()
    
    if (e <= 150 and e%5 == 0) or (e > 150 and and e%1 == 0):
        deeplab.eval()
        # set model flag to eval
        # 1. disable dropout
        # 2. batchnorm behave differs

        with torch.no_grad():
        # stop taking gradient

            #valloss_4 = 0
            #valloss_2 = 0
            valloss_1 = 0

            for v, vbatch in enumerate(validation_loader):
            # iter over validation mini batches

                image_1_val = vbatch['image1_data'].to(device=device, dtype=dtype)
                if get_dimensions(image_1_val) == 4:
                    image_1_val.unsqueeze_(0)
                label_1_val = vbatch['image1_label'].to(device=device, dtype=dtype)
                if get_dimensions(label_1_val) == 4:
                    label_1_val.unsqueeze_(0)
                # move data to device, convert dtype to desirable dtype
                # add one dimension to labels if they are 4D tensors

                out_1_val = deeplab(image_1_val)
                # do the inference

                loss_1 = dice_loss_3(out_1_val, label_1_val)
                # calculate loss

                valloss_1 += loss_1.item()
                # record mini batch loss

            avg_val_loss = (valloss_1 / (v+1))
            outstr = '------- 1st valloss={0:.4f}'\
                .format(avg_val_loss) + '\n'

            logger['validation_1'].append(avg_val_loss)
            #scheduler.step(avg_val_loss)

            print(outstr)
            record.write(outstr)
            record.flush()

            if avg_val_loss < min_val:
                print(avg_val_loss, "less than", min_val)
                min_val = avg_val_loss
                save_1('deeplab_output_8_elu_save', deeplab, optimizer, logger, e, scheduler)
            elif e%10 == 0:
                save_1('deeplab_output_8_elu_save', deeplab, optimizer, logger, e, scheduler)

record.close()

  0%|          | 0/4669 [00:00<?, ?it/s]

Epoch 331 finished ! Training Loss: 0.0780



  0%|          | 1/4669 [14:01<1091:35:25, 841.84s/it]

------- 1st valloss=0.1245

Epoch 332 finished ! Training Loss: 0.0774



  0%|          | 2/4669 [26:45<1061:05:09, 818.49s/it]

------- 1st valloss=0.0832

Epoch 333 finished ! Training Loss: 0.0867



  0%|          | 3/4669 [39:43<1045:07:38, 806.36s/it]

------- 1st valloss=0.0996

Epoch 334 finished ! Training Loss: 0.0821



  0%|          | 4/4669 [52:22<1026:29:08, 792.14s/it]

------- 1st valloss=0.0658

Epoch 335 finished ! Training Loss: 0.0746



  0%|          | 5/4669 [1:05:16<1019:08:36, 786.65s/it]

------- 1st valloss=0.0651

Epoch 336 finished ! Training Loss: 0.0845



  0%|          | 6/4669 [1:18:14<1015:24:14, 783.93s/it]

------- 1st valloss=0.0707

Epoch 337 finished ! Training Loss: 0.0805



  0%|          | 7/4669 [1:30:57<1007:08:28, 777.72s/it]

------- 1st valloss=0.0802

Epoch 338 finished ! Training Loss: 0.0796



  0%|          | 8/4669 [1:43:48<1004:09:22, 775.58s/it]

------- 1st valloss=0.0766

Epoch 339 finished ! Training Loss: 0.0758



  0%|          | 9/4669 [1:56:26<997:22:47, 770.51s/it] 

------- 1st valloss=0.0935

Epoch 340 finished ! Training Loss: 0.0784

------- 1st valloss=0.0797



  0%|          | 10/4669 [2:09:08<993:43:29, 767.85s/it]

Checkpoint 340 saved !
Epoch 341 finished ! Training Loss: 0.0751



  0%|          | 11/4669 [2:21:56<993:46:50, 768.06s/it]

------- 1st valloss=0.0670

Epoch 342 finished ! Training Loss: 0.0742



  0%|          | 12/4669 [2:34:51<995:55:11, 769.88s/it]

------- 1st valloss=0.0775

Epoch 343 finished ! Training Loss: 0.0880



  0%|          | 13/4669 [2:47:36<993:59:30, 768.55s/it]

------- 1st valloss=0.0780

Epoch 344 finished ! Training Loss: 0.0775



  0%|          | 14/4669 [3:00:32<996:49:22, 770.91s/it]

------- 1st valloss=0.0814

Epoch 345 finished ! Training Loss: 0.0740



  0%|          | 15/4669 [3:13:20<995:08:24, 769.77s/it]

------- 1st valloss=0.0970

Epoch 346 finished ! Training Loss: 0.0747



  0%|          | 16/4669 [3:26:14<996:54:13, 771.30s/it]

------- 1st valloss=0.0750

Epoch 347 finished ! Training Loss: 0.0717



  0%|          | 17/4669 [3:39:07<997:04:44, 771.60s/it]

------- 1st valloss=0.0704

Epoch 348 finished ! Training Loss: 0.0709



  0%|          | 18/4669 [3:52:04<998:57:13, 773.22s/it]

------- 1st valloss=0.0694

Epoch 349 finished ! Training Loss: 0.0703



  0%|          | 19/4669 [4:05:00<999:47:38, 774.03s/it]

------- 1st valloss=0.0757

Epoch 350 finished ! Training Loss: 0.0697

------- 1st valloss=0.1190



  0%|          | 20/4669 [4:17:51<998:21:17, 773.09s/it]

Checkpoint 350 saved !
Epoch 351 finished ! Training Loss: 0.0697



  0%|          | 21/4669 [4:30:56<1002:45:17, 776.66s/it]

------- 1st valloss=0.0719

Epoch 352 finished ! Training Loss: 0.0706



  0%|          | 22/4669 [4:43:28<993:16:34, 769.48s/it] 

------- 1st valloss=0.0926

Epoch 353 finished ! Training Loss: 0.0700



  0%|          | 23/4669 [4:56:13<991:22:24, 768.18s/it]

------- 1st valloss=0.0645

Epoch 354 finished ! Training Loss: 0.0757



  1%|          | 24/4669 [5:09:13<995:33:58, 771.59s/it]

------- 1st valloss=0.0865

Epoch 355 finished ! Training Loss: 0.0730



  1%|          | 25/4669 [5:21:57<992:28:51, 769.37s/it]

------- 1st valloss=0.0777

Epoch 356 finished ! Training Loss: 0.0710



  1%|          | 26/4669 [5:34:50<993:42:45, 770.49s/it]

------- 1st valloss=0.1305

Epoch 357 finished ! Training Loss: 0.0962



  1%|          | 27/4669 [5:47:44<994:54:10, 771.57s/it]

------- 1st valloss=0.1864



In [None]:
deeplab.eval()

with torch.no_grad():
    
    bgloss = 0
    bdloss = 0
    bvloss = 0
    
    for v, vbatch in tqdm(enumerate(validation_loader)):
            # move data to device, convert dtype to desirable dtype

        image_1 = vbatch['image1_data'].to(device=device, dtype=dtype)
        label_1 = vbatch['image1_label'].to(device=device, dtype=dtype)

        output = deeplab(image_1)
        # do the inference
        output_numpy = output.cpu().numpy()
        
        
        #out_1 = torch.round(output)
        out_1 = torch.from_numpy((output_numpy == output_numpy.max(axis=1)[:, None]).astype(int)).to(device=device, dtype=dtype)
        loss_1 = dice_loss_3(out_1, label_1)

        bg, bd, bv = dice_loss_3_debug(out_1, label_1)
        # calculate loss
        print(bg.item(), bd.item(), bv.item(), loss_1.item())
        bgloss += bg.item()
        bdloss += bd.item()
        bvloss += bv.item()

    outstr = '------- background loss = {0:.4f}, body loss = {1:.4f}, bv loss = {2:.4f}'\
        .format(bgloss/(v+1), bdloss/(v+1), bvloss/(v+1)) + '\n'
    print(outstr)