{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T19:17:06Z","timestamp":1783192626511,"version":"3.54.6"},"reference-count":42,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,11,1]],"date-time":"2026-11-01T00:00:00Z","timestamp":1793491200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62566063"],"award-info":[{"award-number":["62566063"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Pattern Recognition"],"published-print":{"date-parts":[[2026,11]]},"DOI":"10.1016\/j.patcog.2026.113944","type":"journal-article","created":{"date-parts":[[2026,5,12]],"date-time":"2026-05-12T15:48:26Z","timestamp":1778600906000},"page":"113944","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PD","title":["DocSR-DETR: Enhancing Document Layout Analysis with emphasis on background regions of document images"],"prefix":"10.1016","volume":"179","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-0465-3117","authenticated-orcid":false,"given":"Wenjie","family":"Xu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mayire","family":"Ibrayim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Quan","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Linying","family":"Wang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.patcog.2026.113944_b1","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110314","article-title":"Reading order detection in visually-rich documents with multi-modal layout-aware relation prediction","volume":"150","author":"Qiao","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113944_b2","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2024.110836","article-title":"Detect-order-construct: A tree construction based approach for hierarchical document structure analysis","volume":"156","author":"Wang","year":"2024","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113944_b3","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2022.108660","article-title":"Synthetic document generator for annotation-free layout recognition","volume":"128","author":"Raman","year":"2022","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113944_b4","doi-asserted-by":"crossref","unstructured":"L. Ouyang, Y. Qu, H. Zhou, J. Zhu, R. Zhang, Q. Lin, B. Wang, Z. Zhao, M. Jiang, X. Zhao, et al., Omnidocbench: Benchmarking diverse pdf document parsing with comprehensive annotations, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2025, pp. 24838\u201324848.","DOI":"10.1109\/CVPR52734.2025.02313"},{"key":"10.1016\/j.patcog.2026.113944_b5","doi-asserted-by":"crossref","unstructured":"H. Cheng, P. Zhang, S. Wu, J. Zhang, Q. Zhu, Z. Xie, J. Li, K. Ding, L. Jin, M6doc: A large-scale multi-format, multi-type, multi-layout, multi-language, multi-annotation category dataset for modern document layout analysis, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 15138\u201315147.","DOI":"10.1109\/CVPR52729.2023.01453"},{"key":"10.1016\/j.patcog.2026.113944_b6","doi-asserted-by":"crossref","unstructured":"C. Da, C. Luo, Q. Zheng, C. Yao, Vision grid transformer for document layout analysis, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 19462\u201319472.","DOI":"10.1109\/ICCV51070.2023.01783"},{"issue":"10","key":"10.1016\/j.patcog.2026.113944_b7","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1007\/s11760-025-04422-y","article-title":"An advanced unimodal approach to solving complex document layout analysis tasks","volume":"19","author":"Xu","year":"2025","journal-title":"Signal, Image Video Process."},{"key":"10.1016\/j.patcog.2026.113944_b8","doi-asserted-by":"crossref","unstructured":"X. Hou, M. Liu, S. Zhang, P. Wei, B. Chen, Salience detr: Enhancing detection transformer with hierarchical salience filtering refinement, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 17574\u201317583.","DOI":"10.1109\/CVPR52733.2024.01664"},{"key":"10.1016\/j.patcog.2026.113944_b9","series-title":"European Conference on Computer Vision","first-page":"89","article-title":"Relation detr: Exploring explicit position relation prior for object detection","author":"Hou","year":"2024"},{"key":"10.1016\/j.patcog.2026.113944_b10","doi-asserted-by":"crossref","DOI":"10.1016\/j.patcog.2025.112031","article-title":"Mamba-YOLO: Multi-level adaptive rectangular convolution for document layout analysis","volume":"170","author":"Ma","year":"2026","journal-title":"Pattern Recognit."},{"key":"10.1016\/j.patcog.2026.113944_b11","series-title":"International Conference on Document Analysis and Recognition","first-page":"76","article-title":"Doc-dino: A transformer model for complex logical document layout analysis","author":"Deng","year":"2024"},{"key":"10.1016\/j.patcog.2026.113944_b12","first-page":"1539","volume":"vol. 18","author":"Deng","year":"2024"},{"issue":"3","key":"10.1016\/j.patcog.2026.113944_b13","doi-asserted-by":"crossref","first-page":"457","DOI":"10.1007\/s10032-025-00542-w","article-title":"SlimDoc: lightweight distillation of document transformer models: M. lamott et al.","volume":"28","author":"Lamott","year":"2025","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"key":"10.1016\/j.patcog.2026.113944_b14","unstructured":"A. Gu, T. Dao, Mamba: Linear-time sequence modeling with selective state spaces, in: First Conference on Language Modeling, 2024."},{"key":"10.1016\/j.patcog.2026.113944_b15","series-title":"European Conference on Computer Vision","first-page":"1","article-title":"Yolov9: Learning what you want to learn using programmable gradient information","author":"Wang","year":"2024"},{"key":"10.1016\/j.patcog.2026.113944_b16","doi-asserted-by":"crossref","unstructured":"Y. Huang, T. Lv, L. Cui, Y. Lu, F. Wei, Layoutlmv3: Pre-training for document ai with unified text and image masking, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 4083\u20134091.","DOI":"10.1145\/3503161.3548112"},{"key":"10.1016\/j.patcog.2026.113944_b17","series-title":"Doclayout-yolo: Enhancing document layout analysis through diverse synthetic data and global-to-local adaptive perception","author":"Zhao","year":"2024"},{"key":"10.1016\/j.patcog.2026.113944_b18","doi-asserted-by":"crossref","unstructured":"J. Li, Y. Xu, T. Lv, L. Cui, C. Zhang, F. Wei, Dit: Self-supervised pre-training for document image transformer, in: Proceedings of the 30th ACM International Conference on Multimedia, 2022, pp. 3530\u20133539.","DOI":"10.1145\/3503161.3547911"},{"key":"10.1016\/j.patcog.2026.113944_b19","first-page":"7233","article-title":"M2doc: a multi-modal fusion approach for document layout analysis","volume":"vol. 38","author":"Zhang","year":"2024"},{"key":"10.1016\/j.patcog.2026.113944_b20","series-title":"International Conference on Document Analysis and Recognition","first-page":"115","article-title":"VSR: a unified framework for document layout analysis combining vision, semantics and relations","author":"Zhang","year":"2021"},{"key":"10.1016\/j.patcog.2026.113944_b21","doi-asserted-by":"crossref","unstructured":"Y. Xu, M. Li, L. Cui, S. Huang, F. Wei, M. Zhou, Layoutlm: Pre-training of text and layout for document image understanding, in: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, 2020, pp. 1192\u20131200.","DOI":"10.1145\/3394486.3403172"},{"key":"10.1016\/j.patcog.2026.113944_b22","doi-asserted-by":"crossref","unstructured":"Y. Xu, Y. Xu, T. Lv, L. Cui, F. Wei, G. Wang, Y. Lu, D. Florencio, C. Zhang, W. Che, et al., Layoutlmv2: Multi-modal pre-training for visually-rich document understanding, in: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics and the 11th International Joint Conference on Natural Language Processing (Volume 1: Long Papers), 2021, pp. 2579\u20132591.","DOI":"10.18653\/v1\/2021.acl-long.201"},{"key":"10.1016\/j.patcog.2026.113944_b23","series-title":"Sparse detr: Efficient end-to-end object detection with learnable sparsity","author":"Roh","year":"2021"},{"key":"10.1016\/j.patcog.2026.113944_b24","doi-asserted-by":"crossref","unstructured":"D. Zheng, W. Dong, H. Hu, X. Chen, Y. Wang, Less is more: Focus attention for efficient detr, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 6674\u20136683.","DOI":"10.1109\/ICCV51070.2023.00614"},{"key":"10.1016\/j.patcog.2026.113944_b25","doi-asserted-by":"crossref","unstructured":"D. Jia, Y. Yuan, H. He, X. Wu, H. Yu, W. Lin, L. Sun, C. Zhang, H. Hu, Detrs with hybrid matching, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2023, pp. 19702\u201319712.","DOI":"10.1109\/CVPR52729.2023.01887"},{"key":"10.1016\/j.patcog.2026.113944_b26","doi-asserted-by":"crossref","unstructured":"Q. Chen, X. Chen, J. Wang, S. Zhang, K. Yao, H. Feng, J. Han, E. Ding, G. Zeng, J. Wang, Group detr: Fast detr training with group-wise one-to-many assignment, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 6633\u20136642.","DOI":"10.1109\/ICCV51070.2023.00610"},{"key":"10.1016\/j.patcog.2026.113944_b27","series-title":"International Conference on Intelligent Computing","first-page":"144","article-title":"Lp-detr: Layer-wise progressive relation for object detection","author":"Kang","year":"2025"},{"key":"10.1016\/j.patcog.2026.113944_b28","doi-asserted-by":"crossref","unstructured":"K. He, X. Zhang, S. Ren, J. Sun, Deep residual learning for image recognition, in: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, 2016, pp. 770\u2013778.","DOI":"10.1109\/CVPR.2016.90"},{"key":"10.1016\/j.patcog.2026.113944_b29","doi-asserted-by":"crossref","unstructured":"Z. Liu, Y. Lin, Y. Cao, H. Hu, Y. Wei, Z. Zhang, S. Lin, B. Guo, Swin transformer: Hierarchical vision transformer using shifted windows, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2021, pp. 10012\u201310022.","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"10.1016\/j.patcog.2026.113944_b30","doi-asserted-by":"crossref","unstructured":"X. Cai, Q. Lai, Y. Wang, W. Wang, Z. Sun, Y. Yao, Poly kernel inception network for remote sensing detection, in: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, 2024, pp. 27706\u201327716.","DOI":"10.1109\/CVPR52733.2024.02617"},{"issue":"6","key":"10.1016\/j.patcog.2026.113944_b31","doi-asserted-by":"crossref","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","article-title":"Faster R-CNN: Towards real-time object detection with region proposal networks","volume":"39","author":"Ren","year":"2016","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"10.1016\/j.patcog.2026.113944_b32","series-title":"2019 International Conference on Document Analysis and Recognition","first-page":"1015","article-title":"Publaynet: largest dataset ever for document layout analysis","author":"Zhong","year":"2019"},{"key":"10.1016\/j.patcog.2026.113944_b33","doi-asserted-by":"crossref","unstructured":"B. Pfitzmann, C. Auer, M. Dolfi, A.S. Nassar, P. Staar, Doclaynet: A large human-annotated dataset for document-layout segmentation, in: Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining, 2022, pp. 3743\u20133751.","DOI":"10.1145\/3534678.3539043"},{"key":"10.1016\/j.patcog.2026.113944_b34","doi-asserted-by":"crossref","unstructured":"M. Li, Y. Xu, L. Cui, S. Huang, F. Wei, Z. Li, M. Zhou, Docbank: A benchmark dataset for document layout analysis, in: Proceedings of the 28th International Conference on Computational Linguistics, 2020, pp. 949\u2013960.","DOI":"10.18653\/v1\/2020.coling-main.82"},{"key":"10.1016\/j.patcog.2026.113944_b35","series-title":"2015 13th International Conference on Document Analysis and Recognition","first-page":"991","article-title":"Evaluation of deep convolutional nets for document image classification and retrieval","author":"Harley","year":"2015"},{"key":"10.1016\/j.patcog.2026.113944_b36","series-title":"European Conference on Computer Vision","first-page":"740","article-title":"Microsoft coco: Common objects in context","author":"Lin","year":"2014"},{"key":"10.1016\/j.patcog.2026.113944_b37","article-title":"A method for stochastic optimization","volume":"vol. 5","author":"Kinga","year":"2015"},{"key":"10.1016\/j.patcog.2026.113944_b38","doi-asserted-by":"crossref","unstructured":"M. Ye, L. Ke, S. Li, Y.-W. Tai, C.-K. Tang, M. Danelljan, F. Yu, Cascade-DETR: Delving into high-quality universal object detection, in: Proceedings of the IEEE\/CVF International Conference on Computer Vision, 2023, pp. 6704\u20136714.","DOI":"10.1109\/ICCV51070.2023.00617"},{"key":"10.1016\/j.patcog.2026.113944_b39","doi-asserted-by":"crossref","first-page":"107984","DOI":"10.52202\/079017-3429","article-title":"Yolov10: Real-time end-to-end object detection","volume":"37","author":"Wang","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113944_b40","doi-asserted-by":"crossref","first-page":"103031","DOI":"10.52202\/079017-3273","article-title":"Vmamba: Visual state space model","volume":"37","author":"Liu","year":"2024","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"10.1016\/j.patcog.2026.113944_b41","doi-asserted-by":"crossref","unstructured":"D. Lewis, G. Agam, S. Argamon, O. Frieder, D. Grossman, J. Heard, Building a test collection for complex document information processing, in: Proceedings of the 29th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval, 2006, pp. 665\u2013666.","DOI":"10.1145\/1148170.1148307"},{"key":"10.1016\/j.patcog.2026.113944_b42","series-title":"Surface defect detection of casting with machined surfaces based on natural artificial defects","author":"Wang","year":"2023"}],"container-title":["Pattern Recognition"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032600909X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S003132032600909X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T19:04:03Z","timestamp":1783191843000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S003132032600909X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,11]]},"references-count":42,"alternative-id":["S003132032600909X"],"URL":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113944","relation":{},"ISSN":["0031-3203"],"issn-type":[{"value":"0031-3203","type":"print"}],"subject":[],"published":{"date-parts":[[2026,11]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"DocSR-DETR: Enhancing Document Layout Analysis with emphasis on background regions of document images","name":"articletitle","label":"Article Title"},{"value":"Pattern Recognition","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.patcog.2026.113944","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"113944"}}