{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,12]],"date-time":"2026-06-12T16:08:13Z","timestamp":1781280493214,"version":"3.54.1"},"reference-count":33,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2023,11,19]],"date-time":"2023-11-19T00:00:00Z","timestamp":1700352000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,11,19]],"date-time":"2023-11-19T00:00:00Z","timestamp":1700352000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100001809","name":"the National Natural Science Foundation of China","doi-asserted-by":"crossref","award":["No.62166043"],"award-info":[{"award-number":["No.62166043"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"crossref"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SIViP"],"published-print":{"date-parts":[[2024,3]]},"DOI":"10.1007\/s11760-023-02838-y","type":"journal-article","created":{"date-parts":[[2023,11,19]],"date-time":"2023-11-19T02:01:42Z","timestamp":1700359302000},"page":"1539-1548","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":19,"title":["The YOLO model that still excels in document layout analysis"],"prefix":"10.1007","volume":"18","author":[{"given":"Qilin","family":"Deng","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Mayire","family":"Ibrayim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Askar","family":"Hamdulla","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chunhu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2023,11,19]]},"reference":[{"issue":"105555","key":"2838_CR1","first-page":"2969239","volume":"9199","author":"S Ren","year":"2015","unstructured":"Ren, S., He, K., Girshick, R., et al.: Faster r-cnn: Towards real-time object detection with region proposal networks. Adv. Neural Inform. Process. Syst. 9199(105555), 2969239\u201350 (2015)","journal-title":"Adv. Neural Inform. Process. Syst."},{"key":"2838_CR2","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., et al.: Mask r-cnn[C]\/\/In: Proceedings of the IEEE International Conference on Computer Vision. (2017): 2961-2969","DOI":"10.1109\/ICCV.2017.322"},{"issue":"3","key":"2838_CR3","doi-asserted-by":"publisher","first-page":"370","DOI":"10.1006\/cviu.1998.0684","volume":"70","author":"K Kise","year":"1998","unstructured":"Kise, K., Sato, A., Iwata, M.: Segmentation of page images using the area Voronoi diagram. Comput. Vis. Image Underst. 70(3), 370\u2013382 (1998)","journal-title":"Comput. Vis. Image Underst."},{"key":"2838_CR4","unstructured":"Nagy, George, Seth, Sharad: (1984). Hierarchical representation of optically scanned documents. In: The International Conference on Pattern Recognition. IEEE, 347-349"},{"issue":"29","key":"2838_CR5","first-page":"12021","volume":"20","author":"Jia Yun","year":"2020","unstructured":"Yun, Jia, Xuedong, Tian, Lina, Zuo: A method for analyzing ancient book layout images based on local outlier factors and fluctuation thresholds. Sci. Technol. Eng. 20(29), 12021\u201312027 (2020)","journal-title":"Sci. Technol. Eng."},{"key":"2838_CR6","doi-asserted-by":"crossref","unstructured":"Saha, R., Mondal, A., Jawahar, C.V.: Graphical object detection in document images. In: 2019 International Conference on Document Analysis and Recognition (ICDAR). IEEE: 51-58 (2019)","DOI":"10.1109\/ICDAR.2019.00018"},{"key":"2838_CR7","doi-asserted-by":"publisher","first-page":"738","DOI":"10.1109\/ICDAR.2019.00123","volume":"2019","author":"R Alaasam","year":"2019","unstructured":"Alaasam, R., Kurar, B., El-Sana, J.: Layout analysis on challenging historical Arabic manuscripts using Siamese network. Int. Conf. Document Anal. Recognit. (ICDAR) 2019, 738\u2013742 (2019). https:\/\/doi.org\/10.1109\/ICDAR.2019.00123","journal-title":"Int. Conf. Document Anal. Recognit. (ICDAR)"},{"key":"2838_CR8","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., Ren, S., Sun, J.: Deep residual learning for image recognition. In CVPR, (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"2838_CR9","doi-asserted-by":"publisher","unstructured":"Yang, H., Hsu, W.H.: \u201dVision-Based Layout Detection from Scientific Literature using Recurrent Convolutional Neural Networks,\u201d In: 2020 25th International Conference on Pattern Recognition (ICPR), (2021), pp. 6455-6462, https:\/\/doi.org\/10.1109\/ICPR48806.2021.9412557.","DOI":"10.1109\/ICPR48806.2021.9412557."},{"key":"2838_CR10","doi-asserted-by":"crossref","unstructured":"Lee, Y., Hwang, J., Lee, S., et al.: An energy and GPU-computation efficient backbone network for real-time object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops. (2019): 0-0","DOI":"10.1109\/CVPRW.2019.00103"},{"key":"2838_CR11","doi-asserted-by":"crossref","unstructured":"Huang, Y., et al.: \"LayoutLMv3: Pre-training for Document AI with Unified Text and Image Masking.\" (2022)","DOI":"10.1145\/3503161.3548112"},{"key":"2838_CR12","doi-asserted-by":"crossref","unstructured":"Liu, Zhuang et al. \u201cA ConvNet for the 2020s.\u201d In: 2022 IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR) (2022): 11966-11976","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"2838_CR13","first-page":"15908","volume":"34","author":"K Han","year":"2021","unstructured":"Han, K., Xiao, A., Wu, E., et al.: Transformer in transformer. Adv. Neural. Inf. Process. Syst. 34, 15908\u201315919 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2838_CR14","doi-asserted-by":"crossref","unstructured":"Dai, Jifeng et al. \u201cDeformable Convolutional Networks.\u201d In: 2017 IEEE International Conference on Computer Vision (ICCV) (2017): 764-773","DOI":"10.1109\/ICCV.2017.89"},{"key":"2838_CR15","first-page":"6789","volume":"35","author":"A Goyal","year":"2022","unstructured":"Goyal, A., Bochkovskiy, A., Deng, J., et al.: Non-deep networks. Adv. Neural. Inf. Process. Syst. 35, 6789\u20136801 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2838_CR16","doi-asserted-by":"crossref","unstructured":"Gao, L., et al. \u201cICDAR2017 competition on page object detection,\u201d In: 2017 14th IAPR Int. Conf. Doc. Anal. Recogn. (ICDAR), IEEE, vol. 1, pp. 1417-1422, (2017)","DOI":"10.1109\/ICDAR.2017.231"},{"key":"2838_CR17","doi-asserted-by":"crossref","unstructured":"Zhong, X., Tang, J., Yepes, A.J.: \u201cPubLayNet: largest dataset ever for document layout analysis\u2019.\u2019 In: 2019 Int. Conf. Document Anal Recog. (ICDAR), IEEE, pp. 1015-1022, (2019)","DOI":"10.1109\/ICDAR.2019.00166"},{"key":"2838_CR18","doi-asserted-by":"crossref","unstructured":"Mondal, A., Lipps, P., Jawahar, C.: IIIT-AR-13K: A new dataset for graphical object detection in documents. In: International Workshop on Document Analysis Systems; Springer: Cham, Switzerland, (2020); pp. 216-230","DOI":"10.1007\/978-3-030-57058-3_16"},{"key":"2838_CR19","doi-asserted-by":"crossref","unstructured":"Gao, L., Yi, X., Jiang, Z., Hao, L., Tang, Z.: \u201cICDAR2017 competition on page object detection\u201d. In ICDAR, (2017)","DOI":"10.1109\/ICDAR.2017.231"},{"issue":"18","key":"2838_CR20","doi-asserted-by":"publisher","first-page":"6460","DOI":"10.3390\/app10186460","volume":"10","author":"J Younas","year":"2020","unstructured":"Younas, J., Siddiqui, S.A., Munir, M., et al.: Fi-fo detector: Figure and formula detection using deformable networks. Appl. Sci. 10(18), 6460 (2020)","journal-title":"Appl. Sci."},{"key":"2838_CR21","doi-asserted-by":"publisher","unstructured":"Bi, H., Xu, C., Shi, C., et al.: SRRV: A novel document object detector based on spatial-related relation and vision. IEEE Trans. Multimed. 25, 3788\u20133798 (2023). https:\/\/doi.org\/10.1109\/TMM.2022.3165717","DOI":"10.1109\/TMM.2022.3165717"},{"issue":"1","key":"2838_CR22","first-page":"10","volume":"3","author":"H Zhang","year":"2023","unstructured":"Zhang, H., Xu, C., Shi, C., et al.: HSCA-Net: A hybrid spatial-channel attention network in multiscale feature pyramid for document layout analysis. J. Artif. Intell. Technol. 3(1), 10\u201317 (2023)","journal-title":"J. Artif. Intell. Technol."},{"key":"2838_CR23","doi-asserted-by":"crossref","unstructured":"Li, X.-H., Yin, F., Liu, C.-L.: \u201cPage object detection from pdf document images by deep structured prediction and supervised clustering\u201d. In: 2018 24th International Conference on Pattern Recognition (ICPR), (2018), pp. 3627-3632","DOI":"10.1109\/ICPR.2018.8546073"},{"key":"2838_CR24","doi-asserted-by":"crossref","unstructured":"Li, K., Wigington, C., Tensmeyer, C., Zhao, H., Barmpalios, N., Morariu, V.I., Manjunatha, V., Sun, T., Fu, Y.: \"Cross-domain documentobject detection: Benchmark suite and method\u201d. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 12 915-12 924 (2020)","DOI":"10.1109\/CVPR42600.2020.01293"},{"key":"2838_CR25","doi-asserted-by":"crossref","unstructured":"He, K. et al. \"Masked Autoencoders Are Scalable Vision Learners.\" (2021)","DOI":"10.1109\/CVPR52688.2022.01553"},{"key":"2838_CR26","unstructured":"Gu, J., et al.: UniDoc: Unified Pretraining Framework for Document Understanding. Adv. Neural Inform. Process. Syst. 34, 39\u201350 (2021)"},{"key":"2838_CR27","doi-asserted-by":"crossref","unstructured":"Li, Junlong, Xu, Yiheng, Lv, Tengchao, Cui, Lei, Zhang, Cha, Wei, Furu: Dit: Self-supervised pre-training for docu-ment image transformer. arXiv preprint arXiv:2203.02378,2022","DOI":"10.1145\/3503161.3547911"},{"key":"2838_CR28","unstructured":"Bao, H., Dong, L., Piao, S., et al. Beit: Bert pre-training of image transformers. arXiv preprint arXiv:2106.08254, (2021)"},{"key":"2838_CR29","doi-asserted-by":"crossref","unstructured":"Zhang, P., Li, C., Qiao, L., Cheng, Z., Pu, S., Niu, Y., Wu, F.: VSR:a unified framework for document layout analysis combining vision, semantics and relations [C]\/\/Document Analysis and Recognition\u2013ICDAR 2021: 16th International Conference, Lausanne, Switzerland, September 5\u201310, 2021, Proceedings, Part I 16. Springer International Publishing, pp. 115\u2013130","DOI":"10.1007\/978-3-030-86549-8_8"},{"key":"2838_CR30","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., Doll\u00e1r, P., et al.: Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition. (2017): 1492-1500","DOI":"10.1109\/CVPR.2017.634"},{"key":"2838_CR31","doi-asserted-by":"crossref","unstructured":"Nguyen, P., Ngo, L., Truong, T.: Nguyen, T.T.; Vo, N.D.; Nguyen, K. Page Object Detection with YOLOF. In: Proceedings of the 2021 8th NAFOSTED Conference on Information and Computer Science (NICS), Hanoi, Vietnam, 21-22 December 2021;pp. 205-210","DOI":"10.1109\/NICS54270.2021.9701449"},{"issue":"6","key":"2838_CR32","doi-asserted-by":"publisher","first-page":"176","DOI":"10.3390\/fi14060176","volume":"14","author":"G Kallempudi","year":"2022","unstructured":"Kallempudi, G., Hashmi, K.A., Pagani, A., et al.: Toward semi-supervised graphical object detection in document images. Future Internet 14(6), 176 (2022)","journal-title":"Future Internet"},{"key":"2838_CR33","doi-asserted-by":"crossref","unstructured":"Naik, S., Hashmi, K.A., Pagani, A., et al.: \"Investigating attention mechanism for page object detection in document images\". Appl. Sci. 12(15), 7486 (2022)","DOI":"10.3390\/app12157486"}],"container-title":["Signal, Image and Video Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-023-02838-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s11760-023-02838-y\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s11760-023-02838-y.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,20]],"date-time":"2024-02-20T02:16:08Z","timestamp":1708395368000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s11760-023-02838-y"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,11,19]]},"references-count":33,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2024,3]]}},"alternative-id":["2838"],"URL":"https:\/\/doi.org\/10.1007\/s11760-023-02838-y","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-3268193\/v1","asserted-by":"object"}]},"ISSN":["1863-1703","1863-1711"],"issn-type":[{"value":"1863-1703","type":"print"},{"value":"1863-1711","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,11,19]]},"assertion":[{"value":"16 August 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"16 August 2023","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 October 2023","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"19 November 2023","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Conflict of Interests The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}