open-mmlab · Junjun2016 · Aug 30, 2021 · Jun 17, 2021 · Jun 17, 2021 · Jun 18, 2021
diff --git a/configs/_base_/models/dpt_vit-b16.py b/configs/_base_/models/dpt_vit-b16.py
@@ -0,0 +1,32 @@
+norm_cfg = dict(type='SyncBN', requires_grad=True)
+model = dict(
+    type='EncoderDecoder',
+    pretrained='pretrain/vit-b16_p16_224.pth', # noqa
+    backbone=dict(
+        type='VisionTransformer',
+        img_size=224,
+        embed_dims=768,
+        num_layers=12,
+        num_heads=12,
+        out_indices=(2, 5, 8, 11),
+        final_norm=False,
+        with_cls_token=True,
+        with_cp=True,
+        output_cls_token=True),
+    decode_head=dict(
+        type='DPTHead',
+        in_channels=(768, 768, 768, 768),
+        channels=256,
+        embed_dims=768,
+        post_process_channels=[96, 192, 384, 768],
+        num_classes=150,
+        readout_type='project',
+        input_transform='multiple_select',
+        in_index=(0, 1, 2, 3),
+        norm_cfg=norm_cfg,
+        loss_decode=dict(
+            type='CrossEntropyLoss', use_sigmoid=False, loss_weight=1.0)),
+    auxiliary_head=None,
+    # model training and testing settings
+    train_cfg=dict(),
+    test_cfg=dict(mode='whole'))  # yapf: disable
diff --git a/configs/dpt/README.md b/configs/dpt/README.md
@@ -0,0 +1,40 @@
+# Vision Transformer for Dense Prediction
+
+## Introduction
+
+<!-- [ALGORITHM] -->
+
+```latex
+@article{dosoViTskiy2020,
+  title={An Image is Worth 16x16 Words: Transformers for Image Recognition at Scale},
+  author={DosoViTskiy, Alexey and Beyer, Lucas and Kolesnikov, Alexander and Weissenborn, Dirk and Zhai, Xiaohua and Unterthiner, Thomas and  Dehghani, Mostafa and Minderer, Matthias and Heigold, Georg and Gelly, Sylvain and Uszkoreit, Jakob and Houlsby, Neil},
+  journal={arXiv preprint arXiv:2010.11929},
+  year={2020}
+}
+
+@article{Ranftl2021,
+  author    = {Ren\'{e} Ranftl and Alexey Bochkovskiy and Vladlen Koltun},
+  title     = {Vision Transformers for Dense Prediction},
+  journal   = {ArXiv preprint},
+  year      = {2021},
+}
+```
+
+## How to use ViT pretrain weights
+
+We convert the backbone weights from the pytorch-image-models repo (https://github.com/rwightman/pytorch-image-models) with `tools/model_converters/vit_convert.py`.
+
+You may follow below steps to start segformer training preparation:
+
+1. Download segformer pretrain weights (Suggest put in `pretrain/`);
+2. Run convert script to convert official pretrain weights: `python tools/model_converters/vit_convert.py pretrain/vit_timm.pth pretrain/vit-b16__p16_224.pth`;
+3. Modify `pretrained` of VisionTransformer model config, for example, `pretrained` of `dpt_vit-b16.py` is set to `pretrain/vit-b16_p16_224.pth`;
+
+## Results and models
+
+### ADE20K
+
+| Method  | Backbone | Crop Size | Lr schd | Mem (GB) | Inf time (fps) |  mIoU | mIoU(ms+flip) | config                                                                                                                 | download                                                                                                                                                                                                                                                                                                                               |
+| ------- | -------- | --------- | ------: | -------- | -------------- | ----: | ------------: | ---------------------------------------------------------------------------------------------------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- |
+| DPT | ViT-B | 512x512  | 160000  | 8.09 | 10.41 | 46.97 | 48.34 | [config](https://github.com/open-mmlab/mmsegmentation/blob/master/configs/dpt/dpt_vit-b16_512x512_160k_ade20k.py) | [model](https://download.openmmlab.com/mmsegmentation/v0.5/dpt/dpt_vit-b16_512x512_160k_ade20k/dpt_vit-b16_512x512_160k_ade20k-db31cf52.pth) &#124; [log](https://download.openmmlab.com/mmsegmentation/v0.5/dpt/dpt_vit-b16_512x512_160k_ade20k/dpt_vit-b16_512x512_160k_ade20k-20210809_172025.log.json) |
+| DPT | ViT-L | 512x512  | 160000  | 18.37 | 4.36 | 46.19 | 46.97 | [config](https://github.com/open-mmlab/mmsegmentation/blob/master/configs/dpt/dpt_vit-l16_512x512_160k_ade20k.py) | [model](https://download.openmmlab.com/mmsegmentation/v0.5/dpt/dpt_vit-l16_512x512_160k_ade20k/dpt_vit-l16_512x512_160k_ade20k-7b753ca6.pth) &#124; [log](https://download.openmmlab.com/mmsegmentation/v0.5/dpt/dpt_vit-l16_512x512_160k_ade20k/dpt_vit-l16_512x512_160k_ade20k-20210809_172025.log.json) |
diff --git a/configs/dpt/dpt.yml b/configs/dpt/dpt.yml
@@ -0,0 +1,50 @@
+Collections:
+- Metadata:
+    Training Data:
+    - ADE20K
+  Name: dpt
+Models:
+- Config: configs/dpt/dpt_vit-b16_512x512_160k_ade20k.py
+  In Collection: dpt
+  Metadata:
+    backbone: ViT-B
+    crop size: (512,512)
+    inference time (ms/im):
+    - backend: PyTorch
+      batch size: 1
+      hardware: V100
+      mode: FP32
+      resolution: (512,512)
+      value: 96.06
+    lr schd: 160000
+    memory (GB): 8.09
+  Name: dpt_vit-b16_512x512_160k_ade20k
+  Results:
+    Dataset: ADE20K
+    Metrics:
+      mIoU: 46.97
+      mIoU(ms+flip): 48.34
+    Task: Semantic Segmentation
+  Weights: https://download.openmmlab.com/mmsegmentation/v0.5/dpt/dpt_vit-b16_512x512_160k_ade20k/dpt_vit-b16_512x512_160k_ade20k-db31cf52.pth
+- Config: configs/dpt/dpt_vit-l16_512x512_160k_ade20k.py
+  In Collection: dpt
+  Metadata:
+    backbone: ViT-L
+    crop size: (512,512)
+    inference time (ms/im):
+    - backend: PyTorch
+      batch size: 1
+      hardware: V100
+      mode: FP32
+      resolution: (512,512)
+      value: 229.36
+    lr schd: 160000
+    memory (GB): 18.37
+  Name: dpt_vit-l16_512x512_160k_ade20k
+  Results:
+    Dataset: ADE20K
+    Metrics:
+      mIoU: 46.19
+      mIoU(ms+flip): 46.97
+    Task: Semantic Segmentation
+  Weights: https://download.openmmlab.com/mmsegmentation/v0.5/dpt/dpt_vit-l16_512x512_160k_ade20k/dpt_vit-l16_512x512_160k_ade20k-7b753ca6.pth
diff --git a/configs/dpt/dpt_vit-b16_512x512_160k_ade20k.py b/configs/dpt/dpt_vit-b16_512x512_160k_ade20k.py
@@ -0,0 +1,32 @@
+_base_ = [
+    '../_base_/models/dpt_vit-b16.py', '../_base_/datasets/ade20k.py',
+    '../_base_/default_runtime.py', '../_base_/schedules/schedule_160k.py'
+]
+
+# AdamW optimizer, no weight decay for position embedding & layer norm
+# in backbone
+optimizer = dict(
+    _delete_=True,
+    type='AdamW',
+    lr=0.00006,
+    betas=(0.9, 0.999),
+    weight_decay=0.01,
+    paramwise_cfg=dict(
+        custom_keys={
+            'pos_embed': dict(decay_mult=0.),
+            'cls_token': dict(decay_mult=0.),
+            'norm': dict(decay_mult=0.)
+        }))
+
+lr_config = dict(
+    _delete_=True,
+    policy='poly',
+    warmup='linear',
+    warmup_iters=1500,
+    warmup_ratio=1e-6,
+    power=1.0,
+    min_lr=0.0,
+    by_epoch=False)
+
+# By default, models are trained on 8 GPUs with 2 images per GPU
+data = dict(workers_per_gpu=2)
diff --git a/configs/dpt/dpt_vit-l16_512x512_160k_ade20k.py b/configs/dpt/dpt_vit-l16_512x512_160k_ade20k.py
@@ -0,0 +1,24 @@
+_base_ = './dpt_vit-b16_512x512_160k_ade20k.py'
+
+model = dict(
+    type='EncoderDecoder',
+    pretrained='pretrain/vit-l16_p16_384.pth', # noqa
+    backbone=dict(
+        type='VisionTransformer',
+        img_size=384,
+        embed_dims=1024,
+        num_heads=16,
+        num_layers=24,
+        out_indices=(5, 11, 17, 23),
+        final_norm=False,
+        with_cls_token=True,
+        output_cls_token=True),
+    decode_head=dict(
+        type='DPTHead',
+        in_channels=(1024, 1024, 1024, 1024),
+        channels=256,
+        embed_dims=1024,
+        post_process_channels=[256, 512, 1024, 1024]),
+    # model training and testing settings
+    train_cfg=dict(),
+    test_cfg=dict(mode='whole'))  # yapf: disable
diff --git a/mmseg/models/decode_heads/__init__.py b/mmseg/models/decode_heads/__init__.py
@@ -6,6 +6,7 @@
 from .da_head import DAHead
 from .dm_head import DMHead
 from .dnl_head import DNLHead
+from .dpt_head import DPTHead
 from .ema_head import EMAHead
 from .enc_head import EncHead
 from .fcn_head import FCNHead
@@ -29,5 +30,5 @@
     'UPerHead', 'DepthwiseSeparableASPPHead', 'ANNHead', 'DAHead', 'OCRHead',
     'EncHead', 'DepthwiseSeparableFCNHead', 'FPNHead', 'EMAHead', 'DNLHead',
     'PointHead', 'APCHead', 'DMHead', 'LRASPPHead', 'SETRUPHead',
-    'SETRMLAHead', 'SegformerHead'
+    'SETRMLAHead', 'DPTHead', 'SETRMLAHead', 'SegformerHead'
 ]