{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:28:04Z","timestamp":1781587684261,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2023ZD0121502"],"award-info":[{"award-number":["2023ZD0121502"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["72234005"],"award-info":[{"award-number":["72234005"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Young Scientists Fund of the National Natural Science Foundation of China","award":["72304215"],"award-info":[{"award-number":["72304215"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681609","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"3239-3247","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["DIG: Complex Layout Document Image Generation with Authentic-looking Text for Enhancing Layout Analysis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0005-2180-4184","authenticated-orcid":false,"given":"Dehao","family":"Ying","sequence":"first","affiliation":[{"name":"School of Information Management, Wuhan University, Wuhan, Hubei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6503-4688","authenticated-orcid":false,"given":"Fengchang","family":"Yu","sequence":"additional","affiliation":[{"name":"School of Information Management, Wuhan University, Wuhan, Hubei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7088-9752","authenticated-orcid":false,"given":"Haihua","family":"Chen","sequence":"additional","affiliation":[{"name":"Department of Information Science, University of North Texas, Denton, Texas, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0929-7416","authenticated-orcid":false,"given":"Wei","family":"Lu","sequence":"additional","affiliation":[{"name":"School of Information Management, Wuhan University, Wuhan, Hubei, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Hugging Face 2022. Stable diffusion v1.5 model card. Hugging Face. https: \/\/huggingface.co\/runwayml\/stable-diffusion-v1--5"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2009.271"},{"key":"e_1_3_2_1_3_1","volume-title":"Data augmentation generative adversarial networks. arXiv preprint arXiv:1711.04340","author":"Antoniou Antreas","year":"2017","unstructured":"Antreas Antoniou, Amos Storkey, and Harrison Edwards. 2017. Data augmentation generative adversarial networks. arXiv preprint arXiv:1711.04340 (2017)."},{"key":"e_1_3_2_1_4_1","volume-title":"Synthetic Data from Diffusion Models Improves ImageNet Classification. Transactions on Machine Learning Research","author":"Azizi Shekoofeh","year":"2023","unstructured":"Shekoofeh Azizi, Simon Kornblith, Chitwan Saharia, Mohammad Norouzi, and David J Fleet. 2023. Synthetic Data from Diffusion Models Improves ImageNet Classification. Transactions on Machine Learning Research (2023)."},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00959"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10032-021-00380-6"},{"key":"e_1_3_2_1_7_1","volume-title":"Textdiffuser: Diffusion models as text painters. Advances in Neural Information Processing Systems 36","author":"Chen Jingye","year":"2024","unstructured":"Jingye Chen, Yupan Huang, Tengchao Lv, Lei Cui, Qifeng Chen, and Furu Wei. 2024. Textdiffuser: Diffusion models as text painters. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_8_1","volume-title":"Generative adversarial nets. Advances in neural information processing systems 27","author":"Goodfellow Ian","year":"2014","unstructured":"Ian Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, David Warde-Farley, Sherjil Ozair, Aaron Courville, and Yoshua Bengio. 2014. Generative adversarial nets. Advances in neural information processing systems 27 (2014)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00104"},{"key":"e_1_3_2_1_10_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"He Ruifei","year":"2022","unstructured":"Ruifei He, Shuyang Sun, Xin Yu, Chuhui Xue, Wenqing Zhang, Philip Torr, Song Bai, and XIAOJUAN QI. 2022. IS SYNTHETIC DATA FROM GENERATIVE MODELS READY FOR IMAGE RECOGNITION?. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_1_11_1","volume-title":"Denoising diffusion probabilistic models. Advances in neural information processing systems 33","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems 33 (2020), 6840--6851."},{"key":"e_1_3_2_1_12_1","volume-title":"Generative models as a data source for multiview representation learning. arXiv preprint arXiv:2106.05258","author":"Jahanian Ali","year":"2021","unstructured":"Ali Jahanian, Xavier Puig, Yonglong Tian, and Phillip Isola. 2021. Generative models as a data source for multiview representation learning. arXiv preprint arXiv:2106.05258 (2021)."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00999"},{"key":"e_1_3_2_1_14_1","volume-title":"Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114","author":"Kingma Diederik P","year":"2013","unstructured":"Diederik P Kingma and Max Welling. 2013. Auto-encoding variational bayes. arXiv preprint arXiv:1312.6114 (2013)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02064"},{"key":"e_1_3_2_1_16_1","volume-title":"LayoutGAN: Generating Graphic Layouts with Wireframe Discriminators. In International Conference on Learning Representations.","author":"Li Jianan","year":"2018","unstructured":"Jianan Li, Jimei Yang, Aaron Hertzmann, Jianming Zhang, and Tingfa Xu. 2018. LayoutGAN: Generating Graphic Layouts with Wireframe Discriminators. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_17_1","first-page":"1","article-title":"Grains: Generative recursive autoencoders for indoor scenes","volume":"38","author":"Li Manyi","year":"2019","unstructured":"Manyi Li, Akshay Gadi Patil, Kai Xu, Siddhartha Chaudhuri, Owais Khan, Ariel Shamir, Changhe Tu, Baoquan Chen, Daniel Cohen-Or, and Hao Zhang. 2019. Grains: Generative recursive autoencoders for indoor scenes. ACM Transactions on Graphics (TOG) 38, 2 (2019), 1--16.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"e_1_3_2_1_18_1","volume-title":"DocBank: A benchmark dataset for document layout analysis. arXiv preprint arXiv:2006.01038","author":"Li Minghao","year":"2020","unstructured":"Minghao Li, Yiheng Xu, Lei Cui, Shaohan Huang, Furu Wei, Zhoujun Li, and Ming Zhou. 2020. DocBank: A benchmark dataset for document layout analysis. arXiv preprint arXiv:2006.01038 (2020)."},{"key":"e_1_3_2_1_19_1","volume-title":"Revolutionizing Retrieval-Augmented Generation with Enhanced PDF Structure Recognition. arXiv preprint arXiv:2401.12599","author":"Lin Demiao","year":"2024","unstructured":"Demiao Lin. 2024. Revolutionizing Retrieval-Augmented Generation with Enhanced PDF Structure Recognition. arXiv preprint arXiv:2401.12599 (2024)."},{"key":"e_1_3_2_1_20_1","volume-title":"Thuy Dung Nguyen, and Min-Yen Kan","author":"Luong Minh-Thang","year":"2012","unstructured":"Minh-Thang Luong, Thuy Dung Nguyen, and Min-Yen Kan. 2012. Logical structure recovery in scholarly articles with rich document features. In Multimedia Storage and Retrieval Innovations for Digital Library Systems. IGI Global, 270--292."},{"key":"e_1_3_2_1_21_1","volume-title":"Proceedings, Part XXII 16","author":"Manandhar Dipu","year":"2020","unstructured":"Dipu Manandhar, Dan Ruta, and John Collomosse. 2020. Learning structural similarity of user interface layouts using graph networks. In Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XXII 16. Springer, 730--746."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/34.244677"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1145\/3534678.3539043"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00634"},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/WACV56688.2023.00368"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_27_1","volume-title":"U-net: Convolutional networks for biomedical image segmentation. In Medical image computing and computer-assisted intervention--MICCAI 2015: 18th international conference","author":"Ronneberger Olaf","year":"2015","unstructured":"Olaf Ronneberger, Philipp Fischer, and Thomas Brox. 2015. U-net: Convolutional networks for biomedical image segmentation. In Medical image computing and computer-assisted intervention--MICCAI 2015: 18th international conference, Munich, Germany, October 5--9, 2015, proceedings, part III 18. Springer, 234--241."},{"key":"e_1_3_2_1_28_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 8011--8021","author":"Alahari Karteek","year":"2023","unstructured":"Karteek Alahari, Diane Larlus, and Yannis Kalantidis. 2023. Fake it till you make it: Learning transferable representations from synthetic imagenet clones. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 8011--8021."},{"key":"e_1_3_2_1_29_1","first-page":"25278","article-title":"Laion-5b: An open large-scale dataset for training next generation image-text models","volume":"35","author":"Schuhmann Christoph","year":"2022","unstructured":"Christoph Schuhmann, Romain Beaumont, Richard Vencu, Cade Gordon, Ross Wightman, Mehdi Cherti, Theo Coombes, Aarush Katta, Clayton Mullis, Mitchell Wortsman, et al. 2022. Laion-5b: An open large-scale dataset for training next generation image-text models. Advances in Neural Information Processing Systems 35 (2022), 25278--25294.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_30_1","volume-title":"Proc ICLR","author":"Simonyan K","year":"2015","unstructured":"K Simonyan. 2015. Very deep convolutional networks for large-scale image recognition. Proc ICLR (2015)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.606"},{"key":"e_1_3_2_1_32_1","unstructured":"Brandon Trabucco Kyle Doherty Max Gurinas and Ruslan Salakhutdinov. 2023. Effective Data Augmentation With Diffusion Models. In R0-FoMo: Robustness of Few-shot and Zero-shot Learning in Large Foundation Models."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3306346.3322941","article-title":"Planit: Planning and instantiating indoor scenes with relation graph and spatial prior networks","volume":"38","author":"Wang Kai","year":"2019","unstructured":"Kai Wang, Yu-An Lin, Ben Weissmann, Manolis Savva, Angel X Chang, and Daniel Ritchie. 2019. Planit: Planning and instantiating indoor scenes with relation graph and spatial prior networks. ACM Transactions on Graphics (TOG) 38, 4 (2019), 1--15.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3197517.3201362","article-title":"Deep convolutional priors for indoor scene synthesis","volume":"37","author":"Wang Kai","year":"2018","unstructured":"Kai Wang, Manolis Savva, Angel X Chang, and Daniel Ritchie. 2018. Deep convolutional priors for indoor scene synthesis. ACM Transactions on Graphics (TOG) 37, 4 (2018), 1--14.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"e_1_3_2_1_35_1","volume-title":"Semantic image synthesis via diffusion models. arXiv preprint arXiv:2207.00050","author":"Wang Weilun","year":"2022","unstructured":"Weilun Wang, Jianmin Bao, Wengang Zhou, Dongdong Chen, Dong Chen, Lu Yuan, and Houqiang Li. 2022. Semantic image synthesis via diffusion models. arXiv preprint arXiv:2207.00050 (2022)."},{"key":"e_1_3_2_1_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01370"},{"key":"e_1_3_2_1_37_1","volume-title":"Freemask: Synthetic images with dense annotations make stronger segmentation models. Advances in Neural Information Processing Systems 36","author":"Yang Lihe","year":"2024","unstructured":"Lihe Yang, Xiaogang Xu, Bingyi Kang, Yinghuan Shi, and Hengshuang Zhao. 2024. Freemask: Synthetic images with dense annotations make stronger segmentation models. Advances in Neural Information Processing Systems 36 (2024)."},{"key":"e_1_3_2_1_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.462"},{"key":"e_1_3_2_1_39_1","volume-title":"Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, et al.","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Yuanzhong Xu, Jing Yu Koh, Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, et al. 2022. Scaling Autoregressive Models for Content-Rich Text-to-Image Generation. Transactions on Machine Learning Research (2022)."},{"key":"e_1_3_2_1_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01001"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2019.00166"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681609","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681609","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:49Z","timestamp":1750295869000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681609"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":42,"alternative-id":["10.1145\/3664647.3681609","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681609","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}