{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,13]],"date-time":"2026-05-13T03:18:56Z","timestamp":1778642336278,"version":"3.51.4"},"reference-count":38,"publisher":"Springer Science and Business Media LLC","issue":"10","license":[{"start":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T00:00:00Z","timestamp":1750896000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T00:00:00Z","timestamp":1750896000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["No.62166043"],"award-info":[{"award-number":["No.62166043"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["No.62166043"],"award-info":[{"award-number":["No.62166043"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]},{"name":"Tianshan Talent Training Project-Xinjiang Science and Technology Innovation Team Program","award":["2023TSYCTD0012"],"award-info":[{"award-number":["2023TSYCTD0012"]}]},{"name":"Tianshan Talent Training Project-Xinjiang Science and Technology Innovation Team Program","award":["2023TSYCTD0012"],"award-info":[{"award-number":["2023TSYCTD0012"]}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2025,10]]},"DOI":"10.1007\/s11760-025-04422-y","type":"journal-article","created":{"date-parts":[[2025,6,26]],"date-time":"2025-06-26T10:21:18Z","timestamp":1750933278000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["An Advanced Unimodal Approach to Solving Complex Document Layout Analysis Tasks"],"prefix":"10.1007","volume":"19","author":[{"given":"Wenjie","family":"Xu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mayire","family":"Ibrayim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,6,26]]},"reference":[{"issue":"6","key":"4422_CR1","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3355610","volume":"52","author":"GM Binmakhashen","year":"2019","unstructured":"Binmakhashen, G.M., Mahmoud, S.A.: Document layout analysis: a comprehensive survey. ACM Computing Surveys (CSUR) 52(6), 1\u201336 (2019)","journal-title":"ACM Computing Surveys (CSUR)"},{"key":"4422_CR2","doi-asserted-by":"crossref","unstructured":"Lee, J., Hayashi, H., Ohyama, W., Uchida, S.: Page segmentation using a convolutional neural network with trainable co-occurrence features. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 1023\u20131028 (2019). IEEE","DOI":"10.1109\/ICDAR.2019.00167"},{"key":"4422_CR3","doi-asserted-by":"crossref","unstructured":"Xu, Y., Li, M., Cui, L., Huang, S., Wei, F., Zhou, M.: Layoutlm: Pre-training of text and layout for document image understanding. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining, pp. 1192\u20131200 (2020)","DOI":"10.1145\/3394486.3403172"},{"key":"4422_CR4","doi-asserted-by":"crossref","unstructured":"Xu, Y., Xu, Y., Lv, T., Cui, L., Wei, F., Wang, G., Lu, Y., Florencio, D., Zhang, C., Che, W., et al.: Layoutlmv2: Multi-modal pre-training for visually-rich document understanding. arXiv preprint arXiv:2012.14740 (2020)","DOI":"10.18653\/v1\/2021.acl-long.201"},{"key":"4422_CR5","doi-asserted-by":"crossref","unstructured":"Huang, Y., Lv, T., Cui, L., Lu, Y., Wei, F.: Layoutlmv3: Pre-training for document ai with unified text and image masking. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4083\u20134091 (2022)","DOI":"10.1145\/3503161.3548112"},{"key":"4422_CR6","doi-asserted-by":"crossref","unstructured":"Li, J., Xu, Y., Lv, T., Cui, L., Zhang, C., Wei, F.: Dit: Self-supervised pre-training for document image transformer. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 3530\u20133539 (2022)","DOI":"10.1145\/3503161.3547911"},{"key":"4422_CR7","unstructured":"Zhao, Z., Kang, H., Wang, B., He, C.: Doclayout-yolo: Enhancing document layout analysis through diverse synthetic data and global-to-local adaptive perception. arXiv preprint arXiv:2410.12628 (2024)"},{"key":"4422_CR8","doi-asserted-by":"crossref","unstructured":"Zhang, N., Cheng, H., Chen, J., Jiang, Z., Huang, J., Xue, Y., Jin, L.: M2doc: a multi-modal fusion approach for document layout analysis. In: Proceedings of the AAAI Conference on Artificial Intelligence, vol. 38, pp. 7233\u20137241 (2024)","DOI":"10.1609\/aaai.v38i7.28552"},{"key":"4422_CR9","doi-asserted-by":"crossref","unstructured":"Da, C., Luo, C., Zheng, Q., Yao, C.: Vision grid transformer for document layout analysis. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 19462\u201319472 (2023)","DOI":"10.1109\/ICCV51070.2023.01783"},{"key":"4422_CR10","doi-asserted-by":"crossref","unstructured":"Zhang, P., Li, C., Qiao, L., Cheng, Z., Pu, S., Niu, Y., Wu, F.: Vsr: a unified framework for document layout analysis combining vision, semantics and relations. In: Document Analysis and Recognition\u2013ICDAR 2021: 16th International Conference, Lausanne, Switzerland, September 5\u201310, 2021, Proceedings, Part I 16, pp. 115\u2013130 (2021). Springer","DOI":"10.1007\/978-3-030-86549-8_8"},{"key":"4422_CR11","doi-asserted-by":"crossref","unstructured":"Maity, S., Biswas, S., Manna, S., Banerjee, A., Llad\u00f3s, J., Bhattacharya, S., Pal, U.: Selfdocseg: A self-supervised vision-based approach towards document segmentation. In: International Conference on Document Analysis and Recognition, pp. 342\u2013360 (2023). Springer","DOI":"10.1007\/978-3-031-41676-7_20"},{"key":"4422_CR12","doi-asserted-by":"crossref","unstructured":"Deng, Q., Ibrayim, M., Hamdulla, A., Luo, H., Zhang, C.: Doc-dino: A transformer model for complex logical document layout analysis. In: International Conference on Document Analysis and Recognition, pp. 76\u201389 (2024). Springer","DOI":"10.1007\/978-3-031-70546-5_5"},{"key":"4422_CR13","doi-asserted-by":"crossref","unstructured":"Cheng, H., Zhang, P., Wu, S., Zhang, J., Zhu, Q., Xie, Z., Li, J., Ding, K., Jin, L.: M6doc: A large-scale multi-format, multi-type, multi-layout, multi-language, multi-annotation category dataset for modern document layout analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15138\u201315147 (2023)","DOI":"10.1109\/CVPR52729.2023.01453"},{"issue":"2","key":"4422_CR14","doi-asserted-by":"publisher","first-page":"1539","DOI":"10.1007\/s11760-023-02838-y","volume":"18","author":"Q Deng","year":"2024","unstructured":"Deng, Q., Ibrayim, M., Hamdulla, A., Zhang, C.: The yolo model that still excels in document layout analysis. SIViP 18(2), 1539\u20131548 (2024)","journal-title":"SIViP"},{"key":"4422_CR15","doi-asserted-by":"crossref","unstructured":"Wang, J., Hu, K., Huo, Q.: Dlaformer: An end-to-end transformer for document layout analysis. In: International Conference on Document Analysis and Recognition, pp. 40\u201357 (2024). Springer","DOI":"10.1007\/978-3-031-70546-5_3"},{"issue":"8","key":"4422_CR16","doi-asserted-by":"publisher","first-page":"10479","DOI":"10.1007\/s13369-023-07661-8","volume":"48","author":"S Chowdhury","year":"2023","unstructured":"Chowdhury, S., Soni, B.: Qsfvqa: a time efficient, scalable and optimized vqa framework. Arab. J. Sci. Eng. 48(8), 10479\u201310491 (2023)","journal-title":"Arab. J. Sci. Eng."},{"issue":"6","key":"4422_CR17","doi-asserted-by":"publisher","first-page":"70010","DOI":"10.1111\/coin.70010","volume":"40","author":"S Chowdhury","year":"2024","unstructured":"Chowdhury, S., Soni, B.: Beyond words: Esc-net revolutionizes vqa by elevating visual features and defying language priors. Comput. Intell. 40(6), 70010 (2024)","journal-title":"Comput. Intell."},{"key":"4422_CR18","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2024.112827","volume":"309","author":"S Chowdhury","year":"2025","unstructured":"Chowdhury, S., Soni, B.: R-vqa: a robust visual question answering model. Knowl.-Based Syst. 309, 112827 (2025)","journal-title":"Knowl.-Based Syst."},{"key":"4422_CR19","doi-asserted-by":"publisher","DOI":"10.1016\/j.engappai.2024.109948","volume":"142","author":"S Chowdhury","year":"2025","unstructured":"Chowdhury, S., Soni, B.: Envqa: improving visual question answering model by enriching the visual feature. Eng. Appl. Artif. Intell. 142, 109948 (2025)","journal-title":"Eng. Appl. Artif. Intell."},{"key":"4422_CR20","doi-asserted-by":"publisher","DOI":"10.1016\/j.neucom.2025.129906","volume":"635","author":"S Chowdhury","year":"2025","unstructured":"Chowdhury, S., Soni, B.: Handling language prior and compositional reasoning issues in visual question answering system. Neurocomputing 635, 129906 (2025)","journal-title":"Neurocomputing"},{"key":"4422_CR21","unstructured":"Zhang, H., Li, F., Liu, S., Zhang, L., Su, H., Zhu, J., Ni, L.M., Shum, H.-Y.: Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605 (2022)"},{"key":"4422_CR22","doi-asserted-by":"crossref","unstructured":"Zheng, D., Dong, W., Hu, H., Chen, X., Wang, Y.: Less is more: focus attention for efficient detr. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 6674\u20136683 (2023)","DOI":"10.1109\/ICCV51070.2023.00614"},{"key":"4422_CR23","doi-asserted-by":"crossref","unstructured":"Hou, X., Liu, M., Zhang, S., Wei, P., Chen, B.: Salience detr: enhancing detection transformer with hierarchical salience filtering refinement. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 17574\u201317583 (2024)","DOI":"10.1109\/CVPR52733.2024.01664"},{"key":"4422_CR24","doi-asserted-by":"crossref","unstructured":"Hou, X., Liu, M., Zhang, S., Wei, P., Chen, B., Lan, X.: Relation detr: exploring explicit position relation prior for object detection. In: European Conference on Computer Vision, pp. 89\u2013105 (2024). Springer","DOI":"10.1007\/978-3-031-72973-7_6"},{"key":"4422_CR25","unstructured":"Yao, Z., Ai, J., Li, B., Zhang, C.: Efficient detr: improving end-to-end object detector with dense prior. arXiv preprint arXiv:2104.01318 (2021)"},{"key":"4422_CR26","doi-asserted-by":"crossref","unstructured":"Li, F., Zeng, A., Liu, S., Zhang, H., Li, H., Zhang, L., Ni, L.M.: Lite detr: an interleaved multi-scale encoder for efficient detr. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 18558\u201318567 (2023)","DOI":"10.1109\/CVPR52729.2023.01780"},{"key":"4422_CR27","doi-asserted-by":"crossref","unstructured":"Hu, J., Shen, L., Sun, G.: Squeeze-and-excitation networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7132\u20137141 (2018)","DOI":"10.1109\/CVPR.2018.00745"},{"key":"4422_CR28","unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable detr: Deformable transformers for end-to-end object detection. arXiv preprint arXiv:2010.04159 (2020)"},{"key":"4422_CR29","doi-asserted-by":"crossref","unstructured":"Ding, X., Zhang, X., Ma, N., Han, J., Ding, G., Sun, J.: Repvgg: Making vgg-style convnets great again. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 13733\u201313742 (2021)","DOI":"10.1109\/CVPR46437.2021.01352"},{"key":"4422_CR30","doi-asserted-by":"crossref","unstructured":"Liu, S., Qi, L., Qin, H., Shi, J., Jia, J.: Path aggregation network for instance segmentation. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 8759\u20138768 (2018)","DOI":"10.1109\/CVPR.2018.00913"},{"issue":"6","key":"4422_CR31","doi-asserted-by":"publisher","first-page":"1137","DOI":"10.1109\/TPAMI.2016.2577031","volume":"39","author":"S Ren","year":"2016","unstructured":"Ren, S., He, K., Girshick, R., Sun, J.: Faster r-cnn: towards real-time object detection with region proposal networks. IEEE Trans. Pattern Anal. Mach. Intell. 39(6), 1137\u20131149 (2016)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"4422_CR32","doi-asserted-by":"crossref","unstructured":"Zhong, X., Tang, J., Yepes, A.J.: Publaynet: largest dataset ever for document layout analysis. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), pp. 1015\u20131022 (2019). IEEE","DOI":"10.1109\/ICDAR.2019.00166"},{"key":"4422_CR33","doi-asserted-by":"crossref","unstructured":"Li, M., Xu, Y., Cui, L., Huang, S., Wei, F., Li, Z., Zhou, M.: Docbank: A benchmark dataset for document layout analysis. arXiv preprint arXiv:2006.01038 (2020)","DOI":"10.18653\/v1\/2020.coling-main.82"},{"key":"4422_CR34","doi-asserted-by":"crossref","unstructured":"Pfitzmann, B., Auer, C., Dolfi, M., Nassar, A.S., Staar, P.W.J.: Doclaynet: a large human-annotated dataset for document-layout segmentation. Proceedings of the 28th ACM SIGKDD Conference on Knowledge Discovery and Data Mining (2022)","DOI":"10.1145\/3534678.3539043"},{"key":"4422_CR35","unstructured":"Kinga, D., Adam, J.B., et al.: A method for stochastic optimization. In: International Conference on Learning Representations (ICLR), vol. 5, p. 6 (2015). San Diego, California;"},{"key":"4422_CR36","doi-asserted-by":"crossref","unstructured":"Lewis, D., Agam, G., Argamon, S., Frieder, O., Grossman, D., Heard, J.: Building a test collection for complex document information processing. In: Proceedings of the 29th Annual International ACM SIGIR Conference on Research and Development in Information Retrieval, pp. 665\u2013666 (2006)","DOI":"10.1145\/1148170.1148307"},{"key":"4422_CR37","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. 2016 IEEE Conference on Computer Vision and Pattern Recognition (CVPR), 770\u2013778 (2015)","DOI":"10.1109\/CVPR.2016.90"},{"key":"4422_CR38","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., Cao, Y., Hu, H., Wei, Y., Zhang, Z., Lin, S., Guo, B.: Swin transformer: hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision, pp. 10012\u201310022 (2021)","DOI":"10.1109\/ICCV48922.2021.00986"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04422-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-025-04422-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-025-04422-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,9,6]],"date-time":"2025-09-06T22:48:30Z","timestamp":1757198910000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-025-04422-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,6,26]]},"references-count":38,"journal-issue":{"issue":"10","published-print":{"date-parts":[[2025,10]]}},"alternative-id":["4422"],"URL":"https:\/\/doi.org\/10.1007\/s11760-025-04422-y","relation":{},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,6,26]]},"assertion":[{"value":"31 March 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 June 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 June 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"26 June 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}],"article-number":"794"}}