open-mmlab · KevinNuNu · Mar 23, 2023 · Mar 23, 2023 · Mar 25, 2023 · Mar 25, 2023
diff --git a/.codespellrc b/.codespellrc
@@ -2,4 +2,4 @@
 skip = *.ipynb
 count =
 quiet-level = 3
-ignore-words-list = convertor,convertors,formating,nin,wan,datas,hist,ned
+ignore-words-list = convertor,convertors,formating,nin,wan,datas,hist,ned,ser
diff --git a/configs/re/_base_/datasets/xfund_zh.py b/configs/re/_base_/datasets/xfund_zh.py
@@ -0,0 +1,14 @@
+xfund_zh_re_data_root = 'data/xfund/zh'
+
+xfund_zh_re_train = dict(
+    type='XFUNDDataset',
+    data_root=xfund_zh_re_data_root,
+    ann_file='re_train.json',
+    pipeline=None)
+
+xfund_zh_re_test = dict(
+    type='XFUNDDataset',
+    data_root=xfund_zh_re_data_root,
+    ann_file='re_test.json',
+    test_mode=True,
+    pipeline=None)
diff --git a/configs/ser/_base_/datasets/xfund_zh.py b/configs/ser/_base_/datasets/xfund_zh.py
@@ -0,0 +1,14 @@
+xfund_zh_ser_data_root = 'data/xfund/zh'
+
+xfund_zh_ser_train = dict(
+    type='XFUNDDataset',
+    data_root=xfund_zh_ser_data_root,
+    ann_file='ser_train.json',
+    pipeline=None)
+
+xfund_zh_ser_test = dict(
+    type='XFUNDDataset',
+    data_root=xfund_zh_ser_data_root,
+    ann_file='ser_test.json',
+    test_mode=True,
+    pipeline=None)
diff --git a/dataset_zoo/xfund/de/metafile.yml b/dataset_zoo/xfund/de/metafile.yml
@@ -0,0 +1,41 @@
+Name: 'XFUND'
+Paper:
+  Title: 'XFUND: A Benchmark Dataset for Multilingual Visually Rich Form Understanding'
+  URL: https://aclanthology.org/2022.findings-acl.253
+  Venue: ACL
+  Year: '2022'
+  BibTeX: '@inproceedings{xu-etal-2022-xfund,
+    title = "{XFUND}: A Benchmark Dataset for Multilingual Visually Rich Form Understanding",
+    author = "Xu, Yiheng  and
+      Lv, Tengchao  and
+      Cui, Lei  and
+      Wang, Guoxin  and
+      Lu, Yijuan  and
+      Florencio, Dinei  and
+      Zhang, Cha  and
+      Wei, Furu",
+    booktitle = "Findings of the Association for Computational Linguistics: ACL 2022",
+    month = may,
+    year = "2022",
+    address = "Dublin, Ireland",
+    publisher = "Association for Computational Linguistics",
+    url = "https://aclanthology.org/2022.findings-acl.253",
+    doi = "10.18653/v1/2022.findings-acl.253",
+    pages = "3214--3224",
+    abstract = "Multimodal pre-training with text, layout, and image has achieved SOTA performance for visually rich document understanding tasks recently, which demonstrates the great potential for joint learning across different modalities. However, the existed research work has focused only on the English domain while neglecting the importance of multilingual generalization. In this paper, we introduce a human-annotated multilingual form understanding benchmark dataset named XFUND, which includes form understanding samples in 7 languages (Chinese, Japanese, Spanish, French, Italian, German, Portuguese). Meanwhile, we present LayoutXLM, a multimodal pre-trained model for multilingual document understanding, which aims to bridge the language barriers for visually rich document understanding. Experimental results show that the LayoutXLM model has significantly outperformed the existing SOTA cross-lingual pre-trained models on the XFUND dataset. The XFUND dataset and the pre-trained LayoutXLM model have been publicly available at https://aka.ms/layoutxlm.",
+}'
+Data:
+  Website: https://github.com/doc-analysis/XFUND
+  Language:
+    - Chinese, Japanese, Spanish, French, Italian, German, Portuguese
+  Scene:
+    - Document
+  Granularity:
+    - Word
+  Tasks:
+    - ser
+    - re
+  License:
+    Type: CC BY 4.0
+    Link: https://creativecommons.org/licenses/by/4.0/
+  Format: .json
diff --git a/dataset_zoo/xfund/de/re.py b/dataset_zoo/xfund/de/re.py
@@ -0,0 +1,6 @@
+_base_ = ['ser.py']
+
+_base_.train_preparer.packer.type = 'REPacker'
+_base_.test_preparer.packer.type = 'REPacker'
+
+config_generator = dict(type='XFUNDREConfigGenerator')
diff --git a/dataset_zoo/xfund/de/sample_anno.md b/dataset_zoo/xfund/de/sample_anno.md
@@ -0,0 +1,70 @@
+**Semantic Entity Recognition / Relation Extraction**
+
+```json
+{
+    "lang": "zh",
+    "version": "0.1",
+    "split": "val",
+    "documents": [
+        {
+            "id": "zh_val_0",
+            "uid": "0ac15750a098682aa02b51555f7c49ff43adc0436c325548ba8dba560cde4e7e",
+            "document": [
+                {
+                    "box": [
+                        410,
+                        541,
+                        535,
+                        590
+                    ],
+                    "text": "夏艳辰",
+                    "label": "answer",
+                    "words": [
+                        {
+                            "box": [
+                                413,
+                                541,
+                                447,
+                                587
+                            ],
+                            "text": "夏"
+                        },
+                        {
+                            "box": [
+                                458,
+                                542,
+                                489,
+                                588
+                            ],
+                            "text": "艳"
+                        },
+                        {
+                            "box": [
+                                497,
+                                544,
+                                531,
+                                590
+                            ],
+                            "text": "辰"
+                        }
+                    ],
+                    "linking": [
+                        [
+                            30,
+                            26
+                        ]
+                    ],
+                    "id": 26
+                },
+                // ...
+            ],
+            "img": {
+                "fname": "zh_val_0.jpg",
+                "width": 2480,
+                "height": 3508
+            }
+        },
+        // ...
+    ]
+}
+```
diff --git a/dataset_zoo/xfund/de/ser.py b/dataset_zoo/xfund/de/ser.py
@@ -0,0 +1,60 @@
+lang = 'de'
+data_root = f'data/xfund/{lang}'
+cache_path = 'data/cache'
+
+train_preparer = dict(
+    obtainer=dict(
+        type='NaiveDataObtainer',
+        cache_path=cache_path,
+        files=[
+            dict(
+                url='https://github.com/doc-analysis/XFUND/'
+                f'releases/download/v1.0/{lang}.train.zip',
+                save_name=f'{lang}_train.zip',
+                md5='8c9f949952d227290e22f736cdbe4d29',
+                content=['image'],
+                mapping=[[f'{lang}_train/*.jpg', 'imgs/train']]),
+            dict(
+                url='https://github.com/doc-analysis/XFUND/'
+                f'releases/download/v1.0/{lang}.train.json',
+                save_name=f'{lang}_train.json',
+                md5='3e4b95c7da893bf5a91018445c83ccdd',
+                content=['annotation'],
+                mapping=[[f'{lang}_train.json', 'annotations/train.json']])
+        ]),
+    gatherer=dict(
+        type='MonoGatherer', ann_name='train.json', img_dir='imgs/train'),
+    parser=dict(type='XFUNDAnnParser'),
+    packer=dict(type='SERPacker'),
+    dumper=dict(type='JsonDumper'),
+)
+
+test_preparer = dict(
+    obtainer=dict(
+        type='NaiveDataObtainer',
+        cache_path=cache_path,
+        files=[
+            dict(
+                url='https://github.com/doc-analysis/XFUND/'
+                f'releases/download/v1.0/{lang}.val.zip',
+                save_name=f'{lang}_val.zip',
+                md5='d13d12278d585214183c3cfb949b0e59',
+                content=['image'],
+                mapping=[[f'{lang}_val/*.jpg', 'imgs/test']]),
+            dict(
+                url='https://github.com/doc-analysis/XFUND/'
+                f'releases/download/v1.0/{lang}.val.json',
+                save_name=f'{lang}_val.json',
+                md5='8eaf742f2d19b17f5c0e72da5c7761ef',
+                content=['annotation'],
+                mapping=[[f'{lang}_val.json', 'annotations/test.json']])
+        ]),
+    gatherer=dict(
+        type='MonoGatherer', ann_name='test.json', img_dir='imgs/test'),
+    parser=dict(type='XFUNDAnnParser'),
+    packer=dict(type='SERPacker'),
+    dumper=dict(type='JsonDumper'),
+)
+
+delete = ['annotations'] + [f'{lang}_{split}' for split in ['train', 'val']]
+config_generator = dict(type='XFUNDSERConfigGenerator')
diff --git a/dataset_zoo/xfund/es/metafile.yml b/dataset_zoo/xfund/es/metafile.yml
@@ -0,0 +1,41 @@
+Name: 'XFUND'
+Paper:
+  Title: 'XFUND: A Benchmark Dataset for Multilingual Visually Rich Form Understanding'
+  URL: https://aclanthology.org/2022.findings-acl.253
+  Venue: ACL
+  Year: '2022'
+  BibTeX: '@inproceedings{xu-etal-2022-xfund,
+    title = "{XFUND}: A Benchmark Dataset for Multilingual Visually Rich Form Understanding",
+    author = "Xu, Yiheng  and
+      Lv, Tengchao  and
+      Cui, Lei  and
+      Wang, Guoxin  and
+      Lu, Yijuan  and
+      Florencio, Dinei  and
+      Zhang, Cha  and
+      Wei, Furu",
+    booktitle = "Findings of the Association for Computational Linguistics: ACL 2022",
+    month = may,
+    year = "2022",
+    address = "Dublin, Ireland",
+    publisher = "Association for Computational Linguistics",
+    url = "https://aclanthology.org/2022.findings-acl.253",
+    doi = "10.18653/v1/2022.findings-acl.253",
+    pages = "3214--3224",
+    abstract = "Multimodal pre-training with text, layout, and image has achieved SOTA performance for visually rich document understanding tasks recently, which demonstrates the great potential for joint learning across different modalities. However, the existed research work has focused only on the English domain while neglecting the importance of multilingual generalization. In this paper, we introduce a human-annotated multilingual form understanding benchmark dataset named XFUND, which includes form understanding samples in 7 languages (Chinese, Japanese, Spanish, French, Italian, German, Portuguese). Meanwhile, we present LayoutXLM, a multimodal pre-trained model for multilingual document understanding, which aims to bridge the language barriers for visually rich document understanding. Experimental results show that the LayoutXLM model has significantly outperformed the existing SOTA cross-lingual pre-trained models on the XFUND dataset. The XFUND dataset and the pre-trained LayoutXLM model have been publicly available at https://aka.ms/layoutxlm.",
+}'
+Data:
+  Website: https://github.com/doc-analysis/XFUND
+  Language:
+    - Chinese, Japanese, Spanish, French, Italian, German, Portuguese
+  Scene:
+    - Document
+  Granularity:
+    - Word
+  Tasks:
+    - ser
+    - re
+  License:
+    Type: CC BY 4.0
+    Link: https://creativecommons.org/licenses/by/4.0/
+  Format: .json
diff --git a/dataset_zoo/xfund/es/re.py b/dataset_zoo/xfund/es/re.py
@@ -0,0 +1,6 @@
+_base_ = ['ser.py']
+
+_base_.train_preparer.packer.type = 'REPacker'
+_base_.test_preparer.packer.type = 'REPacker'
+
+config_generator = dict(type='XFUNDREConfigGenerator')
diff --git a/dataset_zoo/xfund/es/sample_anno.md b/dataset_zoo/xfund/es/sample_anno.md
@@ -0,0 +1,70 @@
+**Semantic Entity Recognition / Relation Extraction**
+
+```json
+{
+    "lang": "zh",
+    "version": "0.1",
+    "split": "val",
+    "documents": [
+        {
+            "id": "zh_val_0",
+            "uid": "0ac15750a098682aa02b51555f7c49ff43adc0436c325548ba8dba560cde4e7e",
+            "document": [
+                {
+                    "box": [
+                        410,
+                        541,
+                        535,
+                        590
+                    ],
+                    "text": "夏艳辰",
+                    "label": "answer",
+                    "words": [
+                        {
+                            "box": [
+                                413,
+                                541,
+                                447,
+                                587
+                            ],
+                            "text": "夏"
+                        },
+                        {
+                            "box": [
+                                458,
+                                542,
+                                489,
+                                588
+                            ],
+                            "text": "艳"
+                        },
+                        {
+                            "box": [
+                                497,
+                                544,
+                                531,
+                                590
+                            ],
+                            "text": "辰"
+                        }
+                    ],
+                    "linking": [
+                        [
+                            30,
+                            26
+                        ]
+                    ],
+                    "id": 26
+                },
+                // ...
+            ],
+            "img": {
+                "fname": "zh_val_0.jpg",
+                "width": 2480,
+                "height": 3508
+            }
+        },
+        // ...
+    ]
+}
+```
diff --git a/dataset_zoo/xfund/es/ser.py b/dataset_zoo/xfund/es/ser.py
@@ -0,0 +1,60 @@
+lang = 'es'
+data_root = f'data/xfund/{lang}'
+cache_path = 'data/cache'
+
+train_preparer = dict(
+    obtainer=dict(
+        type='NaiveDataObtainer',
+        cache_path=cache_path,
+        files=[
+            dict(
+                url='https://github.com/doc-analysis/XFUND/'
+                f'releases/download/v1.0/{lang}.train.zip',
+                save_name=f'{lang}_train.zip',
+                md5='0ff89032bc6cb2e7ccba062c71944d03',
+                content=['image'],
+                mapping=[[f'{lang}_train/*.jpg', 'imgs/train']]),
+            dict(
+                url='https://github.com/doc-analysis/XFUND/'
+                f'releases/download/v1.0/{lang}.train.json',
+                save_name=f'{lang}_train.json',
+                md5='b40b43f276c7deaaaa5923d035da2820',
+                content=['annotation'],
+                mapping=[[f'{lang}_train.json', 'annotations/train.json']])
+        ]),
+    gatherer=dict(
+        type='MonoGatherer', ann_name='train.json', img_dir='imgs/train'),
+    parser=dict(type='XFUNDAnnParser'),
+    packer=dict(type='SERPacker'),
+    dumper=dict(type='JsonDumper'),
+)
+
+test_preparer = dict(
+    obtainer=dict(
+        type='NaiveDataObtainer',
+        cache_path=cache_path,
+        files=[
+            dict(
+                url='https://github.com/doc-analysis/XFUND/'
+                f'releases/download/v1.0/{lang}.val.zip',
+                save_name=f'{lang}_val.zip',
+                md5='efad9fb11ee3036bef003b6364a79ac0',
+                content=['image'],
+                mapping=[[f'{lang}_val/*.jpg', 'imgs/test']]),
+            dict(
+                url='https://github.com/doc-analysis/XFUND/'
+                f'releases/download/v1.0/{lang}.val.json',
+                save_name=f'{lang}_val.json',
+                md5='96ffc2057049ba2826a005825b3e7f0d',
+                content=['annotation'],
+                mapping=[[f'{lang}_val.json', 'annotations/test.json']])
+        ]),
+    gatherer=dict(
+        type='MonoGatherer', ann_name='test.json', img_dir='imgs/test'),
+    parser=dict(type='XFUNDAnnParser'),
+    packer=dict(type='SERPacker'),
+    dumper=dict(type='JsonDumper'),
+)
+
+delete = ['annotations'] + [f'{lang}_{split}' for split in ['train', 'val']]
+config_generator = dict(type='XFUNDSERConfigGenerator')