{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,8]],"date-time":"2025-09-08T05:41:23Z","timestamp":1757310083332,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":26,"publisher":"ACM","license":[{"start":{"date-parts":[[2022,3,18]],"date-time":"2022-03-18T00:00:00Z","timestamp":1647561600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"Young Teacher Education Research Project of Fujian","award":["JT180435"],"award-info":[{"award-number":["JT180435"]}]},{"name":"Industry-University Cooperation Project of Fujian Science and Technology Department","award":["2021H6035"],"award-info":[{"award-number":["2021H6035"]}]},{"name":"5th Round of Health and Education Research Program of Fujian Province","award":["2019-WJ-41"],"award-info":[{"award-number":["2019-WJ-41"]}]},{"name":"Natural Science Foundation of China","award":["61773325"],"award-info":[{"award-number":["61773325"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2022,3,18]]},"DOI":"10.1145\/3532213.3532219","type":"proceedings-article","created":{"date-parts":[[2022,7,13]],"date-time":"2022-07-13T13:29:18Z","timestamp":1657718958000},"page":"32-37","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Attention-based Deep Learning Methods for Document Layout Analysis"],"prefix":"10.1145","author":[{"given":"Hao-Yue","family":"Sun","sequence":"first","affiliation":[{"name":"Xiamen University of Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ying","family":"Zhong","sequence":"additional","affiliation":[{"name":"Xiamen University of Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Da-Han","family":"Wang","sequence":"additional","affiliation":[{"name":"Xiamen University of Technology, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2022,7,13]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/LA-CCI47412.2019.9036767"},{"key":"e_1_3_2_1_2_1","first-page":"1","article-title":"Combining visual and textual features for semantic segmentation of historical newspapers","volume":"19","author":"Barman R.","year":"2021","unstructured":"Barman , R. , Ehrmann , M. , Clematide , S. , Oliveira , S.A. , and Kaplan , F. , 2021 . Combining visual and textual features for semantic segmentation of historical newspapers . J. Data Min. Digit. Humanit. 19 , 1 - 26 . http:\/\/dx.doi.org\/10.46298\/JDMDH.6107. 10.46298\/JDMDH.6107 Barman, R., Ehrmann, M., Clematide, S., Oliveira, S.A., and Kaplan, F., 2021. Combining visual and textual features for semantic segmentation of historical newspapers. J. Data Min. Digit. Humanit. 19, 1-26. http:\/\/dx.doi.org\/10.46298\/JDMDH.6107.","journal-title":"J. Data Min. Digit. Humanit."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2020.2976432"},{"key":"e_1_3_2_1_4_1","first-page":"395","article-title":"Historical handwritten document segmentation by using a weighted loss. In Artificial Neural Networks in Pattern Recognition, Springer International Publishing","author":"Capobianco S.","year":"2018","unstructured":"Capobianco , S. , Scommegna , L. , and Marinai , S. , 2018 . Historical handwritten document segmentation by using a weighted loss. In Artificial Neural Networks in Pattern Recognition, Springer International Publishing , Cham , 395 - 406 .http:\/\/dx.doi.org\/10.1007\/978-3-319-99978-4_31. 10.1007\/978-3-319-99978-4_31 Capobianco, S., Scommegna, L., and Marinai, S., 2018. Historical handwritten document segmentation by using a weighted loss. In Artificial Neural Networks in Pattern Recognition, Springer International Publishing, Cham, 395-406.http:\/\/dx.doi.org\/10.1007\/978-3-319-99978-4_31.","journal-title":"Cham"},{"key":"e_1_3_2_1_5_1","first-page":"236","volume-title":"2002 International Conference on Pattern Recognition, Association for Computing Machinery","author":"Cesarini F.","year":"2002","unstructured":"Cesarini , F. , Marinai , S. , Sarti , L. , and Soda , G ., 2002. Trainable table location in document images . In 2002 International Conference on Pattern Recognition, Association for Computing Machinery , New York, NY , 236 - 240 .http:\/\/dx.doi.org\/10.1109\/icpr. 2002 .1047838. 10.1109\/icpr.2002.1047838 Cesarini, F., Marinai, S., Sarti, L., and Soda, G., 2002. Trainable table location in document images. In 2002 International Conference on Pattern Recognition, Association for Computing Machinery, New York, NY, 236-240.http:\/\/dx.doi.org\/10.1109\/icpr.2002.1047838."},{"key":"e_1_3_2_1_6_1","first-page":"1404","volume-title":"ICDAR2017 Competition on recognition of documents with complex layouts - RDCL2017","author":"Clausner C.","year":"2017","unstructured":"Clausner , C. , Antonacopoulos , A. , and Pletschacher , S ., 2017 . ICDAR2017 Competition on recognition of documents with complex layouts - RDCL2017 . In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Kyoto , 1404 - 1410 .http:\/\/dx.doi.org\/10.1109\/ICDAR. 2017 .229. 10.1109\/ICDAR.2017.229 Clausner, C., Antonacopoulos, A., and Pletschacher, S., 2017. ICDAR2017 Competition on recognition of documents with complex layouts - RDCL2017. In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Kyoto, 1404-1410.http:\/\/dx.doi.org\/10.1109\/ICDAR.2017.229."},{"key":"e_1_3_2_1_7_1","first-page":"1521","volume-title":"ICDAR2019 Competition on recognition of documents with complex layouts-RDCL2019","author":"Clausner C.","year":"2019","unstructured":"Clausner , C. , Antonacopoulos , A. , and Pletschacher , S ., 2019 . ICDAR2019 Competition on recognition of documents with complex layouts-RDCL2019 . In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Sydney , 1521 - 1526 .http:\/\/dx.doi.org\/10.1109\/ICDAR. 2019 .00245. 10.1109\/ICDAR.2019.00245 Clausner, C., Antonacopoulos, A., and Pletschacher, S., 2019. ICDAR2019 Competition on recognition of documents with complex layouts-RDCL2019. In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Sydney, 1521-1526.http:\/\/dx.doi.org\/10.1109\/ICDAR.2019.00245."},{"key":"e_1_3_2_1_8_1","volume-title":"ICDAR2017 Competition on the Classification of Medieval Handwritings in Latin Script. In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Kyoto, 1371-1376","author":"Cloppet F.","year":"2017","unstructured":"Cloppet , F. , Eglin , V. , Helias-Baron , M. , Kieu , C. , Vincent , N. , and Stutzmann , D ., 2018 . ICDAR2017 Competition on the Classification of Medieval Handwritings in Latin Script. In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Kyoto, 1371-1376 .http:\/\/dx.doi.org\/10.1109\/ICDAR. 2017 .224. 10.1109\/ICDAR.2017.224 Cloppet, F., Eglin, V., Helias-Baron, M., Kieu, C., Vincent, N., and Stutzmann, D., 2018. ICDAR2017 Competition on the Classification of Medieval Handwritings in Latin Script. In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Kyoto, 1371-1376.http:\/\/dx.doi.org\/10.1109\/ICDAR.2017.224."},{"key":"e_1_3_2_1_9_1","first-page":"854","volume-title":"International Conference on Document Analysis and Recognition, ICDAR, IEEE","author":"Diem M.","year":"2011","unstructured":"Diem , M. , Kleber , F. , and Sablatnig , R ., 2011. Text classification and document layout analysis of paper fragments . In International Conference on Document Analysis and Recognition, ICDAR, IEEE , Beijing , 854 - 858 .http:\/\/dx.doi.org\/10.1109\/ICDAR. 2011 .175. 10.1109\/ICDAR.2011.175 Diem, M., Kleber, F., and Sablatnig, R., 2011. Text classification and document layout analysis of paper fragments. In International Conference on Document Analysis and Recognition, ICDAR, IEEE, Beijing, 854-858.http:\/\/dx.doi.org\/10.1109\/ICDAR.2011.175."},{"key":"e_1_3_2_1_10_1","first-page":"287","volume-title":"12th IAPR International Workshop on Document Analysis Systems, DAS 2016","author":"Hao L.","year":"2016","unstructured":"Hao , L. , Gao , L. , Yi , X. , and Tang , Z ., 2016. A table detection method for PDF documents based on convolutional neural networks . In 12th IAPR International Workshop on Document Analysis Systems, DAS 2016 , Institute of Electrical and Electronics Engineers Inc., Santorini , 287 - 292 .http:\/\/dx.doi.org\/10.1109\/DAS. 2016 .23. 10.1109\/DAS.2016.23 Hao, L., Gao, L., Yi, X., and Tang, Z., 2016. A table detection method for PDF documents based on convolutional neural networks. In 12th IAPR International Workshop on Document Analysis Systems, DAS 2016, Institute of Electrical and Electronics Engineers Inc., Santorini, 287-292.http:\/\/dx.doi.org\/10.1109\/DAS.2016.23."},{"volume-title":"ICIAP 2019: Image Analysis and Processing, Springer Verlag, 292-302","author":"Kavasidis I.","key":"e_1_3_2_1_11_1","unstructured":"Kavasidis , I. , Pino , C. , Palazzo , S. , Rundo , F. , Giordano , D. , Messina , P. , and Spampinato , C ., 2019. A saliency-based convolutional neural network for table and chart detection in digitized documents . In ICIAP 2019: Image Analysis and Processing, Springer Verlag, 292-302 .http:\/\/dx.doi.org\/10.1007\/978-3-030-30645-8_27. 10.1007\/978-3-030-30645-8_27 Kavasidis, I., Pino, C., Palazzo, S., Rundo, F., Giordano, D., Messina, P., and Spampinato, C., 2019. A saliency-based convolutional neural network for table and chart detection in digitized documents. In ICIAP 2019: Image Analysis and Processing, Springer Verlag, 292-302.http:\/\/dx.doi.org\/10.1007\/978-3-030-30645-8_27."},{"key":"e_1_3_2_1_12_1","first-page":"665","volume-title":"Twenty-Ninth Annual International ACM SIGIR Conference on Research and Development in Information Retrieval, Association for Computing Machinery","author":"Lewis D.","unstructured":"Lewis , D. , Agam , G. , Argamon , S. , Frieder , O. , Grossman , D. , and Heard , J ., 2006. Building a test collection for complex document information processing . In Twenty-Ninth Annual International ACM SIGIR Conference on Research and Development in Information Retrieval, Association for Computing Machinery , New York, NY , 665 - 666 .http:\/\/dx.doi.org\/10.1145\/1148170.1148307. 10.1145\/1148170.1148307 Lewis, D., Agam, G., Argamon, S., Frieder, O., Grossman, D., and Heard, J., 2006. Building a test collection for complex document information processing. In Twenty-Ninth Annual International ACM SIGIR Conference on Research and Development in Information Retrieval, Association for Computing Machinery, New York, NY, 665-666.http:\/\/dx.doi.org\/10.1145\/1148170.1148307."},{"key":"e_1_3_2_1_13_1","first-page":"1918","volume-title":"12th Language Resources and Evaluation Conference, European Language Resources Association","author":"Li M.","year":"2020","unstructured":"Li , M. , Cui , L. , Huang , S. , Wei , F. , Zhou , M. , and Li , Z ., 2020. TableBank: Table benchmark for image-based table detection and recognition . In 12th Language Resources and Evaluation Conference, European Language Resources Association , Marseille, France , 1918 - 1925 . Retrieved from https:\/\/aclanthology.org\/ 2020 .lrec-1.236. Li, M., Cui, L., Huang, S., Wei, F., Zhou, M., and Li, Z., 2020. TableBank: Table benchmark for image-based table detection and recognition. In 12th Language Resources and Evaluation Conference, European Language Resources Association, Marseille, France, 1918-1925. Retrieved from https:\/\/aclanthology.org\/2020.lrec-1.236."},{"key":"#cr-split#-e_1_3_2_1_14_1.1","doi-asserted-by":"crossref","unstructured":"Li M. Xu Y. Cui L. Huang S. Wei F. Li Z. and Zhou M. 2021. DocBank: A benchmark dataset for document layout analysis. arXiv Preprint arXiv:2006.01038. http:\/\/dx.doi.org\/10.18653\/V1\/2020.COLING-MAIN.82. 10.18653\/V1","DOI":"10.18653\/v1\/2020.coling-main.82"},{"key":"#cr-split#-e_1_3_2_1_14_1.2","doi-asserted-by":"crossref","unstructured":"Li M. Xu Y. Cui L. Huang S. Wei F. Li Z. and Zhou M. 2021. DocBank: A benchmark dataset for document layout analysis. arXiv Preprint arXiv:2006.01038. http:\/\/dx.doi.org\/10.18653\/V1\/2020.COLING-MAIN.82.","DOI":"10.18653\/v1\/2020.coling-main.82"},{"key":"e_1_3_2_1_15_1","volume-title":"ICDAR2017 Competition on Document Image Binarization (DIBCO 2017). In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Kyoto, 1395-1403","author":"Pratikakis I.","year":"2017","unstructured":"Pratikakis , I. , Zagoris , K. , Barlas , G. , and Gatos , B ., 2017 . ICDAR2017 Competition on Document Image Binarization (DIBCO 2017). In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Kyoto, 1395-1403 .http:\/\/dx.doi.org\/10.1109\/ICDAR. 2017 .228. 10.1109\/ICDAR.2017.228 Pratikakis, I., Zagoris, K., Barlas, G., and Gatos, B., 2017. ICDAR2017 Competition on Document Image Binarization (DIBCO 2017). In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Kyoto, 1395-1403.http:\/\/dx.doi.org\/10.1109\/ICDAR.2017.228."},{"key":"#cr-split#-e_1_3_2_1_16_1.1","doi-asserted-by":"crossref","unstructured":"Quir\u00f3s L. Toselli A.H. and Vidal E. 2019. Multi-task layout analysis of handwritten musical scores. In IbPRIA 2019: Pattern Recognition and Image Analysis Springer Cham 123-134.http:\/\/dx.doi.org\/10.1007\/978-3-030-31321-0_11. 10.1007\/978-3-030-31321-0_11","DOI":"10.1007\/978-3-030-31321-0_11"},{"key":"#cr-split#-e_1_3_2_1_16_1.2","doi-asserted-by":"crossref","unstructured":"Quir\u00f3s L. Toselli A.H. and Vidal E. 2019. Multi-task layout analysis of handwritten musical scores. In IbPRIA 2019: Pattern Recognition and Image Analysis Springer Cham 123-134.http:\/\/dx.doi.org\/10.1007\/978-3-030-31321-0_11.","DOI":"10.1007\/978-3-030-31321-0_11"},{"key":"e_1_3_2_1_17_1","unstructured":"Ran B. and Boyce D. 2012. Modeling dynamic transportation networks: An intelligent transportation system oriented approach. Springer Science & Business Media Berlin.  Ran B. and Boyce D. 2012. Modeling dynamic transportation networks: An intelligent transportation system oriented approach. Springer Science & Business Media Berlin."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2010.194"},{"key":"e_1_3_2_1_19_1","first-page":"3464","article-title":"Visual detection with context for document layout analysis. In 2019 Conference on Empirical Methods in Natural Language Processing and 9th International Joint Conference on Natural Language Processing, Proceedings of the Conference, Association for Computational Linguistics","author":"Soto C.X.","year":"2020","unstructured":"Soto , C.X. and Yoo , S. , 2020 . Visual detection with context for document layout analysis. In 2019 Conference on Empirical Methods in Natural Language Processing and 9th International Joint Conference on Natural Language Processing, Proceedings of the Conference, Association for Computational Linguistics , Hong Kong , 3464 - 3470 .http:\/\/dx.doi.org\/10.18653\/V1\/D19-1348. 10.18653\/V1 Soto, C.X. and Yoo, S., 2020. Visual detection with context for document layout analysis. In 2019 Conference on Empirical Methods in Natural Language Processing and 9th International Joint Conference on Natural Language Processing, Proceedings of the Conference, Association for Computational Linguistics, Hong Kong, 3464-3470.http:\/\/dx.doi.org\/10.18653\/V1\/D19-1348.","journal-title":"Hong Kong"},{"key":"e_1_3_2_1_20_1","first-page":"1220","volume-title":"International Conference on Document Analysis and Recognition, ICDAR, IEEE","author":"Wei H.","year":"2013","unstructured":"Wei , H. , Baechler , M. , Slimane , F. , and Ingold , R ., 2013. Evaluation of SVM, MLP and GMM classifiers for layout analysis of historical documents . In International Conference on Document Analysis and Recognition, ICDAR, IEEE , Washington, DC , 1220 - 1224 .http:\/\/dx.doi.org\/10.1109\/ICDAR. 2013 .247. 10.1109\/ICDAR.2013.247 Wei, H., Baechler, M., Slimane, F., and Ingold, R., 2013. Evaluation of SVM, MLP and GMM classifiers for layout analysis of historical documents. In International Conference on Document Analysis and Recognition, ICDAR, IEEE, Washington, DC, 1220-1224.http:\/\/dx.doi.org\/10.1109\/ICDAR.2013.247."},{"key":"e_1_3_2_1_21_1","volume-title":"International Conference on Frontiers in Handwriting Recognition, ICFHR, Institute of Electrical and Electronics Engineers Inc., Hersonissos, 87-92","author":"Wei H.","year":"2014","unstructured":"Wei , H. , Chen , K. , Ingold , R. , and Liwicki , M ., 2014. Hybrid feature selection for historical document layout analysis . In International Conference on Frontiers in Handwriting Recognition, ICFHR, Institute of Electrical and Electronics Engineers Inc., Hersonissos, 87-92 .http:\/\/dx.doi.org\/10.1109\/ICFHR. 2014 .22. 10.1109\/ICFHR.2014.22 Wei, H., Chen, K., Ingold, R., and Liwicki, M., 2014. Hybrid feature selection for historical document layout analysis. In International Conference on Frontiers in Handwriting Recognition, ICFHR, Institute of Electrical and Electronics Engineers Inc., Hersonissos, 87-92.http:\/\/dx.doi.org\/10.1109\/ICFHR.2014.22."},{"key":"e_1_3_2_1_22_1","first-page":"1192","volume-title":"ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, Association for Computing Machinery","author":"Xu Y.","unstructured":"Xu , Y. , Li , M. , Cui , L. , Huang , S. , Wei , F. , and Zhou , M ., 2020. LayoutLM: Pre-training of text and layout for document image understanding . In ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, Association for Computing Machinery , New York, NY , 1192 - 1200 .http:\/\/dx.doi.org\/10.1145\/3394486.3403172. 10.1145\/3394486.3403172 Xu, Y., Li, M., Cui, L., Huang, S., Wei, F., and Zhou, M., 2020. LayoutLM: Pre-training of text and layout for document image understanding. In ACM SIGKDD International Conference on Knowledge Discovery and Data Mining, Association for Computing Machinery, New York, NY, 1192-1200.http:\/\/dx.doi.org\/10.1145\/3394486.3403172."},{"volume-title":"2nd Indian International Conference on Artificial Intelligence, DBLP Computer Science Bibliography, Pune, 1773-1785","author":"Yildiz B.","key":"e_1_3_2_1_23_1","unstructured":"Yildiz , B. , Kaiser , K. , and Miksch , S ., Pdf2table: A method to extract tfable information from PDF files . In 2nd Indian International Conference on Artificial Intelligence, DBLP Computer Science Bibliography, Pune, 1773-1785 . Retrieved from https:\/\/citeseerx.ist.psu.edu\/viewdoc\/download?doi=10.1.1.724.7272&rep=rep1&type=pdf. Yildiz, B., Kaiser, K., and Miksch, S., Pdf2table: A method to extract tfable information from PDF files. In 2nd Indian International Conference on Artificial Intelligence, DBLP Computer Science Bibliography, Pune, 1773-1785. Retrieved from https:\/\/citeseerx.ist.psu.edu\/viewdoc\/download?doi=10.1.1.724.7272&rep=rep1&type=pdf."},{"key":"e_1_3_2_1_24_1","first-page":"1015","volume-title":"International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society","author":"Zhong X.","year":"2019","unstructured":"Zhong , X. , Tang , J. , and Yepes , A.J ., 2019. PubLayNet: Largest dataset ever for document layout analysis . In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society , Sydney , 1015 - 1022 .http:\/\/dx.doi.org\/10.1109\/ICDAR. 2019 .00166. 10.1109\/ICDAR.2019.00166 Zhong, X., Tang, J., and Yepes, A.J., 2019. PubLayNet: Largest dataset ever for document layout analysis. In International Conference on Document Analysis and Recognition, ICDAR, IEEE Computer Society, Sydney, 1015-1022.http:\/\/dx.doi.org\/10.1109\/ICDAR.2019.00166."}],"event":{"name":"ICCAI '22: 2022 8th International Conference on Computing and Artificial Intelligence","acronym":"ICCAI '22","location":"Tianjin China"},"container-title":["Proceedings of the 8th International Conference on Computing and Artificial Intelligence"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3532213.3532219","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3532213.3532219","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:09:43Z","timestamp":1750183783000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3532213.3532219"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,3,18]]},"references-count":26,"alternative-id":["10.1145\/3532213.3532219","10.1145\/3532213"],"URL":"https:\/\/doi.org\/10.1145\/3532213.3532219","relation":{},"subject":[],"published":{"date-parts":[[2022,3,18]]},"assertion":[{"value":"2022-07-13","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}