{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T07:40:11Z","timestamp":1755848411449,"version":"3.44.0"},"publisher-location":"New York, NY, USA","reference-count":29,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,12,23]],"date-time":"2022-12-23T00:00:00Z","timestamp":1671753600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,12,23]]},"DOI":"10.1145\/3579654.3579767","type":"proceedings-article","created":{"date-parts":[[2023,3,14]],"date-time":"2023-03-14T16:09:40Z","timestamp":1678810180000},"page":"1-6","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["VSLayout: Visual-Semantic Representation Learning For Document Layout Analysis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5004-4851","authenticated-orcid":false,"given":"Shangrong","family":"Wang","sequence":"first","affiliation":[{"name":"School of Computer Science, Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3566-5857","authenticated-orcid":false,"given":"Jing","family":"Jiang","sequence":"additional","affiliation":[{"name":"Department of Communication Engineering, Beijing Union University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8430-1892","authenticated-orcid":false,"given":"Yanjun","family":"Jiang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3185-5100","authenticated-orcid":false,"given":"Xuesong","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Computer Science, Beijing University of Posts and Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,3,14]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2015.7333914"},{"key":"e_1_3_2_1_2_1","unstructured":"Kai Chen Jiaqi Wang Jiangmiao Pang Yuhang Cao Yu Xiong Xiaoxiao Li Shuyang Sun Wansen Feng Ziwei Liu Jiarui Xu Zheng Zhang Dazhi Cheng Chenchen Zhu Tianheng Cheng Qijie Zhao Buyu Li Xin Lu Rui Zhu Yue Wu Jifeng Dai Jingdong Wang Jianping Shi Wanli Ouyang Chen\u00a0Change Loy and Dahua Lin. 2019. MMDetection: Open MMLab Detection Toolbox and Benchmark. arXiv preprint arXiv:1906.07155(2019)."},{"key":"e_1_3_2_1_3_1","unstructured":"Liang-Chieh Chen George Papandreou Florian Schroff and Hartwig Adam. 2017. Rethinking atrous convolution for semantic image segmentation. arXiv preprint arXiv:1706.05587(2017)."},{"key":"e_1_3_2_1_4_1","volume-title":"Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805(2018).","author":"Devlin Jacob","year":"2018","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2018. Bert: Pre-training of deep bidirectional transformers for language understanding. arXiv preprint arXiv:1810.04805(2018)."},{"key":"e_1_3_2_1_5_1","volume-title":"Pp-ocr: A practical ultra lightweight ocr system. arXiv preprint arXiv:2009.09941(2020).","author":"Du Yuning","year":"2020","unstructured":"Yuning Du, Chenxia Li, Ruoyu Guo, Xiaoting Yin, Weiwei Liu, Jun Zhou, Yifan Bai, Zilin Yu, Yehua Yang, Qingqing Dang, 2020. Pp-ocr: A practical ultra lightweight ocr system. arXiv preprint arXiv:2009.09941(2020)."},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.213"},{"key":"e_1_3_2_1_7_1","unstructured":"Kaiming He Xinlei Chen Saining Xie Yanghao Li Piotr Doll\u00e1r and Ross Girshick. 2021. Masked autoencoders are scalable vision learners. arXiv preprint arXiv:2111.06377(2021)."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.322"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"e_1_3_2_1_10_1","volume-title":"Combining visual and textual features for semantic segmentation of historical newspapers. Journal of Data Mining & Digital Humanities","author":"Kaplan Fr\u00e9d\u00e9ric","year":"2021","unstructured":"Fr\u00e9d\u00e9ric Kaplan, Sofia\u00a0Ares Oliveira, Simon Clematide, Maud Ehrmann, and Rapha\u00ebl Barman. 2021. Combining visual and textual features for semantic segmentation of historical newspapers. Journal of Data Mining & Digital Humanities (2021)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01293"},{"key":"e_1_3_2_1_12_1","volume-title":"VTLayout: Fusion of Visual and Text Features for Document Layout Analysis. In Pacific Rim International Conference on Artificial Intelligence. Springer, 308\u2013322","author":"Li Shoubin","year":"2021","unstructured":"Shoubin Li, Xuyan Ma, Shuaiqun Pan, Jun Hu, Lin Shi, and Qing Wang. 2021. VTLayout: Fusion of Visual and Text Features for Document Layout Analysis. In Pacific Rim International Conference on Artificial Intelligence. Springer, 308\u2013322."},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.106"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.170"},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298965"},{"key":"e_1_3_2_1_17_1","unstructured":"Tomas Mikolov Kai Chen Greg Corrado and Jeffrey Dean. 2013. Efficient estimation of word representations in vector space. arXiv preprint arXiv:1301.3781(2013)."},{"key":"e_1_3_2_1_18_1","volume-title":"Distributed representations of words and phrases and their compositionality. Advances in neural information processing systems 26","author":"Mikolov Tomas","year":"2013","unstructured":"Tomas Mikolov, Ilya Sutskever, Kai Chen, Greg\u00a0S Corrado, and Jeff Dean. 2013. Distributed representations of words and phrases and their compositionality. Advances in neural information processing systems 26 (2013)."},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"crossref","unstructured":"Nils Reimers and Iryna Gurevych. 2020. Making monolingual sentence embeddings multilingual using knowledge distillation. arXiv preprint arXiv:2004.09813(2020).","DOI":"10.18653\/v1\/2020.emnlp-main.365"},{"key":"e_1_3_2_1_20_1","volume-title":"Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems 28","author":"Ren Shaoqing","year":"2015","unstructured":"Shaoqing Ren, Kaiming He, Ross Girshick, and Jian Sun. 2015. Faster r-cnn: Towards real-time object detection with region proposal networks. Advances in neural information processing systems 28 (2015)."},{"key":"e_1_3_2_1_21_1","volume-title":"Jacob Carlson, and Weining Li.","author":"Shen Zejiang","year":"2021","unstructured":"Zejiang Shen, Ruochen Zhang, Melissa Dell, Benjamin Charles\u00a0Germain Lee, Jacob Carlson, and Weining Li. 2021. LayoutParser: A Unified Toolkit for Deep Learning Based Document Image Analysis. arXiv preprint arXiv:2103.15348(2021)."},{"key":"e_1_3_2_1_22_1","unstructured":"Carlos Soto. 2019. Visual detection with context for document layout analysis. Technical Report. Brookhaven National Lab.(BNL) Upton NY (United States)."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D19-1348"},{"key":"e_1_3_2_1_24_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan\u00a0N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/DAS.2018.39"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.462"},{"key":"e_1_3_2_1_27_1","unstructured":"Fisher Yu and Vladlen Koltun. 2015. Multi-scale context aggregation by dilated convolutions. arXiv preprint arXiv:1511.07122(2015)."},{"key":"e_1_3_2_1_28_1","volume-title":"Semantics and Relations. In International Conference on Document Analysis and Recognition. Springer, 115\u2013130","author":"Zhang Peng","year":"2021","unstructured":"Peng Zhang, Can Li, Liang Qiao, Zhanzhan Cheng, Shiliang Pu, Yi Niu, and Fei Wu. 2021. VSR: A Unified Framework for Document Layout Analysis combining Vision, Semantics and Relations. In International Conference on Document Analysis and Recognition. Springer, 115\u2013130."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2019.00166"}],"event":{"name":"ACAI 2022: 2022 5th International Conference on Algorithms, Computing and Artificial Intelligence","acronym":"ACAI 2022","location":"Sanya China"},"container-title":["Proceedings of the 2022 5th International Conference on Algorithms, Computing and Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3579654.3579767","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3579654.3579767","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,22]],"date-time":"2025-08-22T07:00:49Z","timestamp":1755846049000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3579654.3579767"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,12,23]]},"references-count":29,"alternative-id":["10.1145\/3579654.3579767","10.1145\/3579654"],"URL":"https:\/\/doi.org\/10.1145\/3579654.3579767","relation":{},"subject":[],"published":{"date-parts":[[2022,12,23]]},"assertion":[{"value":"2023-03-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}