{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,4]],"date-time":"2026-05-04T13:13:07Z","timestamp":1777900387170,"version":"3.51.4"},"reference-count":61,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T00:00:00Z","timestamp":1732406400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T00:00:00Z","timestamp":1732406400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["IJDAR"],"published-print":{"date-parts":[[2025,12]]},"DOI":"10.1007\/s10032-024-00508-4","type":"journal-article","created":{"date-parts":[[2024,11,24]],"date-time":"2024-11-24T06:58:23Z","timestamp":1732431503000},"page":"519-538","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":4,"title":["Document image layout detection from scientific literature using combined ConvNext and cascade mask RCNN networks"],"prefix":"10.1007","volume":"28","author":[{"given":"Qinjun","family":"Qiu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yang","family":"Wang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mengqi","family":"Hao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiandong","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Weijie","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Liufeng","family":"Tao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhong","family":"Xie","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,11,24]]},"reference":[{"issue":"7612","key":"508_CR1","doi-asserted-by":"publisher","first-page":"457","DOI":"10.1038\/nj7612-457a","volume":"535","author":"E Landhuis","year":"2016","unstructured":"Landhuis, E.: Scientific literature: Information overload. Nature 535(7612), 457\u2013458 (2016)","journal-title":"Nature"},{"key":"508_CR2","doi-asserted-by":"crossref","unstructured":"Yang, H., Aguirre, C.A., et al.: Pipelines for procedural information extraction from scientific literature: towards recipes using machine learning and data science. In: 2019 International conference on document analysis and recognition workshops (ICDARW), IEEE (2019).","DOI":"10.1109\/ICDARW.2019.10037"},{"key":"508_CR3","unstructured":"Torres-Salinas, D.: Daily growth rate of scientific production on COVID-19. Analysis in databases and open access repositories. arXiv preprint arXiv:2004.06721 (2020)."},{"key":"508_CR4","unstructured":"Lebourgeois, F., Bublinski, Z., et al.: A fast and efficient method for extracting text paragraphs and graphics from unconstrained documents. In: 11th IAPR International Conference on Pattern Recognition. Vol. II. Conference B: Pattern Recognition Methodology and Systems, IEEE Computer Society (1992)"},{"key":"508_CR5","unstructured":"Yang, H.: Deep integrative information extraction from scientific literature, Kansas State University (2022)."},{"issue":"21","key":"508_CR6","doi-asserted-by":"publisher","first-page":"9436","DOI":"10.1021\/acs.chemmater.7b03500","volume":"29","author":"E Kim","year":"2017","unstructured":"Kim, E., Huang, K., et al.: Materials synthesis insights from scientific literature via text extraction and machine learning. Chem. Mater. 29(21), 9436\u20139444 (2017)","journal-title":"Chem. Mater."},{"key":"508_CR7","first-page":"197","volume":"5010","author":"S Mao","year":"2003","unstructured":"Mao, S., Rosenfeld, A., et al.: Document structure analysis algorithms: a literature survey. Doc. Recognit. Retriev. X 5010, 197\u2013207 (2003)","journal-title":"Doc. Recognit. Retriev. X"},{"issue":"5","key":"508_CR8","first-page":"1415","volume":"1","author":"B Kruatrachue","year":"2007","unstructured":"Kruatrachue, B., Moongfangklang, N., et al.: Fast document segmentation using contourand XY cut technique. Int. J. Comput. Inform. Eng. 1(5), 1415\u20131417 (2007)","journal-title":"Int. J. Comput. Inform. Eng."},{"key":"508_CR9","doi-asserted-by":"crossref","unstructured":"Shilman, M., Liang, P., et al.: Learning nongenerative grammatical models for document analysis. In: Tenth IEEE International Conference on Computer Vision (ICCV'05) Volume 1, IEEE (2005)","DOI":"10.1109\/ICCV.2005.140"},{"key":"508_CR10","doi-asserted-by":"crossref","unstructured":"Wei, H., Baechler, M., et al.: Evaluation of svm, mlp and gmm classifiers for layout analysis of historical documents. In: 2013 12th International Conference on Document Analysis and Recognition, IEEE (2013)","DOI":"10.1109\/ICDAR.2013.247"},{"key":"508_CR11","doi-asserted-by":"crossref","unstructured":"Yang, H., Hsu, W.: Automatic metadata information extraction from scientific literature using deep neural networks. In: Fourteenth International Conference on Machine Vision (ICMV 2021), SPIE (2022)","DOI":"10.1117\/12.2623554"},{"issue":"7","key":"508_CR12","doi-asserted-by":"publisher","first-page":"644","DOI":"10.5626\/JOK.2019.46.7.644","volume":"46","author":"S Kim","year":"2019","unstructured":"Kim, S., Ji, S., et al.: Metadata extraction based on deep learning from academic paper in PDF. J. KIISE 46(7), 644\u2013652 (2019)","journal-title":"J. KIISE"},{"key":"508_CR13","doi-asserted-by":"crossref","unstructured":"Zhong, X., Tang, J., et al.: Publaynet: largest dataset ever for document layout analysis. In: 2019 International Conference on Document Analysis and Recognition (ICDAR), IEEE (2019)","DOI":"10.1109\/ICDAR.2019.00166"},{"key":"508_CR14","doi-asserted-by":"crossref","unstructured":"Li, M., Xu, Y., et al.: DocBank: A benchmark dataset for document layout analysis. arXiv preprint arXiv:2006.01038 (2020)","DOI":"10.18653\/v1\/2020.coling-main.82"},{"key":"508_CR15","doi-asserted-by":"crossref","unstructured":"Melinda, L., Ghanapuram, R., et al.: Document layout analysis using multigaussian fitting. 2017 14th IAPR International conference on document analysis and recognition (ICDAR), IEEE (2017)","DOI":"10.1109\/ICDAR.2017.127"},{"key":"508_CR16","doi-asserted-by":"crossref","unstructured":"Xu, Y., Li, M., et al.: Layoutlm: Pre-training of text and layout for document image understanding. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining (2020)","DOI":"10.1145\/3394486.3403172"},{"key":"508_CR17","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1007\/s10032-015-0249-8","volume":"18","author":"D Tkaczyk","year":"2015","unstructured":"Tkaczyk, D., Szostek, P., et al.: CERMINE: automatic extraction of structured metadata from scientific literature. Int. J. Doc. Anal. Recognit. (IJDAR) 18, 317\u2013335 (2015)","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"key":"508_CR18","doi-asserted-by":"crossref","unstructured":"Lopez, P.: GROBID: Combining automatic bibliographic data recognition and term extraction for scholarship publications. In: Proceedings on Research and Advanced Technology for Digital Libraries: 13th European Conference, ECDL 2009, Corfu, Greece, Springer, (2009)","DOI":"10.1007\/978-3-642-04346-8_62"},{"key":"508_CR19","doi-asserted-by":"crossref","unstructured":"Liu, Z., Mao, H., et al.: A convnet for the 2020s. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2022)","DOI":"10.1109\/CVPR52688.2022.01167"},{"key":"508_CR20","doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., et al.: Mask R-CNN IEEE International Conference on Computer Vision (ICCV) (2017)","DOI":"10.1109\/ICCV.2017.322"},{"key":"508_CR21","doi-asserted-by":"crossref","unstructured":"Cai, Z., Vasconcelos, N.: Cascade r-cnn: delving into high quality object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (2018)","DOI":"10.1109\/CVPR.2018.00644"},{"key":"508_CR22","doi-asserted-by":"crossref","unstructured":"Soto, C., Yoo, S.: Visual detection with context for document layout analysis. In: Proceedings of the 2019 Conference on Empirical Methods in Natural Language Processing and the 9th International Joint Conference on Natural Language Processing (EMNLP-IJCNLP) (2019)","DOI":"10.18653\/v1\/D19-1348"},{"key":"508_CR23","doi-asserted-by":"crossref","unstructured":"Tkaczyk, D., Szostek, P., et al.: GROTOAP2-the methodology of creating a large ground truth dataset of scientific articles. In: D-Lib Magazine 20 (11\/12) (2014)","DOI":"10.1045\/november14-tkaczyk"},{"key":"508_CR24","unstructured":"Ha, J., Haralick, R.M., et al.: Document page decomposition by the bounding-box project. In: Proceedings of 3rd International Conference on Document Analysis and Recognition, IEEE (1995)"},{"issue":"02","key":"508_CR25","doi-asserted-by":"publisher","first-page":"345","DOI":"10.1142\/S0219467801000219","volume":"1","author":"A Amin","year":"2001","unstructured":"Amin, A., Shiu, R.: Page segmentation and classification utilizing bottom-up approach. Int. J. Image Graph. 1(02), 345\u2013361 (2001)","journal-title":"Int. J. Image Graph."},{"key":"508_CR26","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2022.108917","volume":"206","author":"X Yin","year":"2023","unstructured":"Yin, X., Wu, S., et al.: Reversible data hiding in JPEG document images based on zero coefficients embedding. Signal Process. 206, 108917 (2023)","journal-title":"Signal Process."},{"issue":"3","key":"508_CR27","doi-asserted-by":"publisher","first-page":"273","DOI":"10.1109\/34.584106","volume":"19","author":"A Simon","year":"1997","unstructured":"Simon, A., Pret, J., et al.: A fast algorithm for bottom-up document layout analysis. IEEE Trans. Pattern Anal. Mach. Intell. 19(3), 273\u2013277 (1997)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"508_CR28","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.108810","volume":"130","author":"S Suh","year":"2022","unstructured":"Suh, S., Kim, J., et al.: Two-stage generative adversarial networks for binarization of color document images. Pattern Recognit. 130, 108810 (2022)","journal-title":"Pattern Recognit."},{"key":"508_CR29","doi-asserted-by":"publisher","first-page":"99","DOI":"10.1016\/j.eswa.2017.05.030","volume":"85","author":"TA Tran","year":"2017","unstructured":"Tran, T.A., Oh, K., et al.: A robust system for document layout analysis using multilevel homogeneity structure. Expert Syst. Appl. 85, 99\u2013113 (2017)","journal-title":"Expert Syst. Appl."},{"key":"508_CR30","doi-asserted-by":"crossref","unstructured":"Beel, J., Gipp, B., et al.: Docear: An academic literature suite for searching, organizing and creating academic literature. In: Proceedings of the 11th Annual International ACM\/IEEE Joint Conference on Digital Libraries (2011)","DOI":"10.1145\/1998076.1998188"},{"key":"508_CR31","doi-asserted-by":"crossref","unstructured":"Kieninger, T., Dengel, A.: The t-recs table recognition and analysis system. In: Document Analysis Systems: Theory and Practice: Third IAPR Workshop, DAS\u201998 Nagano, Japan, Selected Papers 3, Springer, (1999)","DOI":"10.1007\/3-540-48172-9_21"},{"key":"508_CR32","doi-asserted-by":"crossref","unstructured":"Tsai, C., Kundu, G., et al.: Concept-based analysis of scientific literature. In: Proceedings of the 22nd ACM International Conference on Information & Knowledge Management (2013)","DOI":"10.1145\/2505515.2505613"},{"key":"508_CR33","doi-asserted-by":"publisher","first-page":"31","DOI":"10.1016\/j.patrec.2022.12.024","volume":"166","author":"M Boillet","year":"2023","unstructured":"Boillet, M., Kermorvant, C., et al.: Confidence estimation for object detection in document images. Pattern Recognit. Lett. 166, 31\u201337 (2023)","journal-title":"Pattern Recognit. Lett."},{"key":"508_CR34","doi-asserted-by":"publisher","DOI":"10.1007\/s10032-024-00461-2","author":"A Gemelli","year":"2024","unstructured":"Gemelli, A., et al.: Datasets and annotations for layout analysis of scientific articles. Int. J. Doc. Anal. Recognit. (IJDAR) (2024). https:\/\/doi.org\/10.1007\/s10032-024-00461-2","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"key":"508_CR35","doi-asserted-by":"publisher","DOI":"10.1016\/j.compeleceng.2022.107770","volume":"99","author":"G Jaiswal","year":"2022","unstructured":"Jaiswal, G., Sharma, A., et al.: Deep feature extraction for document forgery detection with convolutional autoencoders. Comput. Electr. Eng. 99, 107770 (2022)","journal-title":"Comput. Electr. Eng."},{"key":"508_CR36","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2021.107882","volume":"115","author":"Y Li","year":"2021","unstructured":"Li, Y., Zhang, P., et al.: Few-shot prototype alignment regularization network for document image layout segementation. Pattern Recognit. 115, 107882 (2021)","journal-title":"Pattern Recognit."},{"issue":"3","key":"508_CR37","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2023.103339","volume":"60","author":"X Wu","year":"2023","unstructured":"Wu, X., Ma, T., et al.: DRFN: A unified framework for complex document layout analysis. Inf. Process. Manage. 60(3), 103339 (2023)","journal-title":"Inf. Process. Manage."},{"key":"508_CR38","doi-asserted-by":"publisher","first-page":"436","DOI":"10.1016\/j.ins.2021.07.020","volume":"577","author":"X Wu","year":"2021","unstructured":"Wu, X., Zheng, Y., et al.: Document image layout analysis via explicit edge embedding network. Inf. Sci. 577, 436\u2013448 (2021)","journal-title":"Inf. Sci."},{"key":"508_CR39","doi-asserted-by":"crossref","unstructured":"Lin, T., Maire, M., et al.: Microsoft coco: Common objects in context. In: Computer Vision\u2014ECCV 2014: 13th European Conference, Zurich, Switzerland, Proceedings, Part V 13, Springer (2014)","DOI":"10.1007\/978-3-319-10602-1_48"},{"issue":"3","key":"508_CR40","doi-asserted-by":"publisher","first-page":"285","DOI":"10.1007\/s10032-019-00332-1","volume":"22","author":"T Gr\u00fcning","year":"2019","unstructured":"Gr\u00fcning, T., Leifert, G., et al.: A two-stage method for text line detection in historical documents. Int. J. Doc. Anal. Recognit. (IJDAR) 22(3), 285\u2013302 (2019)","journal-title":"Int. J. Doc. Anal. Recognit. (IJDAR)"},{"key":"508_CR41","doi-asserted-by":"crossref","unstructured":"Yang, H., Hsu, W.H.: Vision-based layout detection from scientific literature using recurrent convolutional neural networks. In: 2020 25th International Conference on Pattern Recognition (ICPR), IEEE (2021)","DOI":"10.1109\/ICPR48806.2021.9412557"},{"key":"508_CR42","doi-asserted-by":"publisher","DOI":"10.1016\/j.knosys.2023.111080","volume":"282","author":"B Xiao","year":"2023","unstructured":"Xiao, B., Simsek, M., et al.: Table detection for visually rich document images. Knowl.-Based Syst. 282, 111080 (2023)","journal-title":"Knowl.-Based Syst."},{"key":"508_CR43","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2023.110212","volume":"148","author":"K Hu","year":"2024","unstructured":"Hu, K., Zhong, Z., et al.: Mathematical formula detection in document images: A new dataset and a new approach. Pattern Recognit. 148, 110212 (2024)","journal-title":"Pattern Recognit."},{"key":"508_CR44","unstructured":"Dosovitskiy, A., Beyer, L., et al.: An image is worth 16x16 words: Transformers for image recognition at scale. arXiv preprint arXiv:2010.11929 (2020)."},{"key":"508_CR45","unstructured":"Ren, S., He, K., et al.: Faster r-cnn: Towards real-time object detection with region proposal networks. Adv. Neural Inf. Process. Syst 28 (2015)"},{"key":"508_CR46","unstructured":"Loshchilov, I., Hutter, F.: Decoupled weight decay regularization. arXiv preprint arXiv:1711.05101 (2017)"},{"key":"508_CR47","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2023.110136","volume":"136","author":"L Liu","year":"2023","unstructured":"Liu, L., Lu, Y., et al.: End-to-end learning of representations for instance-level document image retrieval. Appl. Soft Comput. 136, 110136 (2023)","journal-title":"Appl. Soft Comput."},{"key":"508_CR48","doi-asserted-by":"crossref","unstructured":"He, K., Zhang, X., et al.: Deep residual learning for image recognition. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (2016)","DOI":"10.1109\/CVPR.2016.90"},{"key":"508_CR49","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. arXiv preprint arXiv:1412.6980 (2014)"},{"key":"508_CR50","doi-asserted-by":"crossref","unstructured":"Liu, Z., Lin, Y., et al.: Swin transformer: Hierarchical vision transformer using shifted windows. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision (2021)","DOI":"10.1109\/ICCV48922.2021.00986"},{"key":"508_CR51","doi-asserted-by":"crossref","unstructured":"Xie, S., Girshick, R., et al.: Aggregated residual transformations for deep neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (2017)","DOI":"10.1109\/CVPR.2017.634"},{"key":"508_CR52","unstructured":"Hendrycks, D., Gimpel, K.: Gaussian error linear units (gelus). arXiv preprint arXiv:1606.08415 (2016)"},{"key":"508_CR53","unstructured":"Nair, V., Hinton, G.E.: Rectified linear units improve restricted boltzmann machines. In: Proceedings of the 27th International Conference on Machine Learning (ICML-10) (2010)"},{"key":"508_CR54","unstructured":"Ba, J.L., Kiros, J.R., et al.: Layer normalization. arXiv preprint arXiv:1607.06450 (2016)"},{"key":"508_CR55","unstructured":"Ioffe, S.: Batch renormalization: towards reducing minibatch dependence in batch-normalized models. Adv Neural Inf Process Syst 30 (2017)"},{"key":"508_CR56","unstructured":"Wu, Y., Johnson, J.: \"Rethinking\" batch\" in batchnorm. arXiv preprint arXiv:2105.07576 (2021)"},{"key":"508_CR57","doi-asserted-by":"crossref","unstructured":"Lin, T., Doll\u00e1r, P., et al.: Feature pyramid networks for object detection. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition (2017).","DOI":"10.1109\/CVPR.2017.106"},{"key":"508_CR58","unstructured":"Chen, K., Wang, J., et al.: MMDetection: Open mmlab detection toolbox and benchmark. arXiv preprint arXiv:1906.07155 (2019)"},{"key":"508_CR59","doi-asserted-by":"crossref","unstructured":"Dai, J., Qi, H., et al.: Deformable convolutional networks. In: Proceedings of the IEEE International Conference on Computer Vision (2017)","DOI":"10.1109\/ICCV.2017.89"},{"key":"508_CR60","doi-asserted-by":"crossref","unstructured":"Cao, Y., Xu, J., et al.: Gcnet: Non-local networks meet squeeze-excitation networks and beyond. In: Proceedings of the IEEE\/CVF International Conference on Computer Vision Workshops (2019)","DOI":"10.1109\/ICCVW.2019.00246"},{"key":"508_CR61","doi-asserted-by":"crossref","unstructured":"Zhang, H., Wu, C., et al.: Resnest: Split-attention networks. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (2022)","DOI":"10.1109\/CVPRW56347.2022.00309"}],"container-title":["International Journal on Document Analysis and Recognition (IJDAR)"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-024-00508-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10032-024-00508-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10032-024-00508-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,28]],"date-time":"2025-11-28T06:12:57Z","timestamp":1764310377000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10032-024-00508-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,24]]},"references-count":61,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2025,12]]}},"alternative-id":["508"],"URL":"https:\/\/doi.org\/10.1007\/s10032-024-00508-4","relation":{},"ISSN":["1433-2833","1433-2825"],"issn-type":[{"value":"1433-2833","type":"print"},{"value":"1433-2825","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,11,24]]},"assertion":[{"value":"7 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"13 October 2024","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 October 2024","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"24 November 2024","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declaration"}},{"value":"The authors declare that they have no known competing financial interests or personal relationships that could have appeared to influence the work reported in this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}