{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,13]],"date-time":"2025-09-13T16:20:57Z","timestamp":1757780457439,"version":"3.40.3"},"publisher-location":"Cham","reference-count":33,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031705458"},{"type":"electronic","value":"9783031705465"}],"license":[{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,1,1]],"date-time":"2024-01-01T00:00:00Z","timestamp":1704067200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024]]},"DOI":"10.1007\/978-3-031-70546-5_5","type":"book-chapter","created":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T05:02:47Z","timestamp":1725944567000},"page":"76-89","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":2,"title":["Doc-DINO: A Transformer Model for\u00a0Complex Logical Document Layout Analysis"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0008-5232-0176","authenticated-orcid":false,"given":"Qilin","family":"Deng","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8766-0647","authenticated-orcid":false,"given":"Mayire","family":"Ibrayim","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2321-308X","authenticated-orcid":false,"given":"Askar","family":"Hamdulla","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-9829-6203","authenticated-orcid":false,"given":"Hailong","family":"Luo","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2564-0998","authenticated-orcid":false,"given":"Chunhu","family":"Zhang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2024,9,11]]},"reference":[{"doi-asserted-by":"publisher","unstructured":"Zhang, C., Ibrayim , M., Hamdulla, A.: A methodological study of document layout analysis. In: 2022 International Conference on Virtual Reality, Human-Computer Interaction and Artificial Intelligence (VRHCIAI), Changsha, China, pp. 12\u201317 (2022). https:\/\/doi.org\/10.1109\/VRHCIAI57205.2022.00009","key":"5_CR1","DOI":"10.1109\/VRHCIAI57205.2022.00009"},{"doi-asserted-by":"crossref","unstructured":"Lee, J., Hayashi, H., Ohyama, W., Uchida, S.: Page segmentation using a convolutional neural network with trainable co-occurrence features. In: ICDAR, pp. 1023\u20131028 (2019). 2, 3","key":"5_CR2","DOI":"10.1109\/ICDAR.2019.00167"},{"doi-asserted-by":"crossref","unstructured":"Yang, X., Yumer, E., Asente, P., Kraley, M., Kifer, D., Lee Giles, C.: Learning to extract semantic structure from documents using multimodal fully convolutional neural net-works. In: IEEE Conference on Computer Vision and Pattern Recognition, pp. 5315\u20135324 (2017)","key":"5_CR3","DOI":"10.1109\/CVPR.2017.462"},{"doi-asserted-by":"crossref","unstructured":"Kise, K., Sato, A., Iwata, M.: Segmentation of page images using the area Voronoi diagram. Comput. Vis. Image Understanding 70(3), 370\u2013382 (1998)","key":"5_CR4","DOI":"10.1006\/cviu.1998.0684"},{"issue":"6","key":"5_CR5","doi-asserted-by":"publisher","first-page":"647","DOI":"10.1147\/rd.266.0647","volume":"26","author":"KY Wong","year":"1982","unstructured":"Wong, K.Y., Casey, R.G., Wahl, F.M.: Document analysis system. IBM J. Res. Dev. 26(6), 647\u2013656 (1982)","journal-title":"IBM J. Res. Dev."},{"issue":"29","key":"5_CR6","first-page":"12021","volume":"20","author":"J Yun","year":"2020","unstructured":"Yun, J., Xuedong, T., Lina, Z.: A method for analyzing ancient book layout images based on local outlier factors and fluctuation thresholds. Sci. Technol. Eng. 20(29), 12021\u201312027 (2020)","journal-title":"Sci. Technol. Eng."},{"unstructured":"Ren, S., He, K., Girshick, R., et al.: Faster R-CNN: towards real-time object detection with region proposal networks. In: Advances in Neural Information Processing Systems (2015). 28","key":"5_CR7"},{"doi-asserted-by":"crossref","unstructured":"He, K., Gkioxari, G., Doll\u00e1r, P., et al.: Mask r-cnn. In: Proceedings of the IEEE International Conference on Computer Vision, pp. 2961\u20132969 (2017)","key":"5_CR8","DOI":"10.1109\/ICCV.2017.322"},{"unstructured":"Vaswani, A., Shazeer, N., Parmar, N., et al.: Attention is all you need. Advances in neural information processing systems (2017). 30","key":"5_CR9"},{"key":"5_CR10","first-page":"15908","volume":"34","author":"K Han","year":"2021","unstructured":"Han, K., Xiao, A., Wu, E., et al.: Transformer in transformer. Adv. Neural. Inf. Process. Syst. 34, 15908\u201315919 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"unstructured":"Zhang, H., et al.: Dino: Detr with improved denoising anchor boxes for end-to-end object detection. arXiv preprint arXiv:2203.03605 (2022)","key":"5_CR11"},{"doi-asserted-by":"crossref","unstructured":"Saha, R., Mondal, A., Jawahar, C.V.: Graphical object detection in document images. In: 2019 Inter national Conference on Document Analysis and Recognition (ICDAR), Sydney, NSW, Australia, pp. 51\u201358. IEEE (2019)","key":"5_CR12","DOI":"10.1109\/ICDAR.2019.00018"},{"doi-asserted-by":"publisher","unstructured":"Yang, H., Hsu, W.H.: Vision-based layout detection from scientific literature using recurrent convolutional neural networks. In: 2020 25th International Conference on Pattern Recognition (ICPR), pp. 6455\u20136462 (2021). https:\/\/doi.org\/10.1109\/ICPR48806.2021.9412557.","key":"5_CR13","DOI":"10.1109\/ICPR48806.2021.9412557."},{"doi-asserted-by":"crossref","unstructured":"Lee, Y., Hwang, J., Lee, S., et al.: An energy and GPU-computation efficient backbone network for real-time object detection. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops (2019)","key":"5_CR14","DOI":"10.1109\/CVPRW.2019.00103"},{"doi-asserted-by":"publisher","unstructured":"Minouei, M., Soheili, M.R., Stricker, D.: Document layout analysis with an enhanced object detector. In: 2021 5th International Conference on Pattern Recognition and Image Analysis (IPRIA), Kashan, Iran, pp. 1\u20135 (2021). https:\/\/doi.org\/10.1109\/IPRIA53572.2021.9483509.","key":"5_CR15","DOI":"10.1109\/IPRIA53572.2021.9483509."},{"doi-asserted-by":"crossref","unstructured":"Zhong, X., Tang, J., Yepes, A.J.: PubLayNet: largest dataset ever for document layout analysis. In: 2019 Int. Conf. Document Anal Recog. (ICDAR), pp. 1015\u20131022. IEEE (2019)","key":"5_CR16","DOI":"10.1109\/ICDAR.2019.00166"},{"doi-asserted-by":"crossref","unstructured":"Zhang, P., Li, C., Qiao, L., Cheng, Z., Pu, S., Niu, Y., Wu, F.: Vsr: a unified framework for document layout analysis combining vision, semantics and relations (2021)","key":"5_CR17","DOI":"10.1007\/978-3-030-86549-8_8"},{"doi-asserted-by":"publisher","unstructured":"Zhong, Z., et al.: A hybrid approach to document layout analysis for heterogeneous document images. In: International Conference on Document Analysis and Recognition. Springer, Cham (2023). https:\/\/doi.org\/10.1007\/978-3-031-41734-4_12","key":"5_CR18","DOI":"10.1007\/978-3-031-41734-4_12"},{"doi-asserted-by":"crossref","unstructured":"Wang, X., Girshick, R., Gupta, A., et al.: Non-local neural networks. In: Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition, pp. 7794\u20137803 (2018)","key":"5_CR19","DOI":"10.1109\/CVPR.2018.00813"},{"unstructured":"Ge, C., Ding, X., Tong, Z., et al.: Advancing Vision Transformers with Group-Mix Attention (2023). arXiv preprint arXiv:2311.15157","key":"5_CR20"},{"doi-asserted-by":"crossref","unstructured":"Cheng, H., Zhang, P., Wu, S., et al.: M6Doc: a large-scale multi-format, multi-type, multi-layout, multi-language, multi-annotation category dataset for modern document layout analysis. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 15138\u201315147 (2023)","key":"5_CR21","DOI":"10.1109\/CVPR52729.2023.01453"},{"doi-asserted-by":"publisher","unstructured":"Cheng, H., Jian, C., Wu, S., et al.: SCUT-CAB: a new benchmark dataset of ancient Chinese books with complex layouts for document layout analysis. In: International Conference on Frontiers in Handwriting Recognition, pp. 436\u2013451. Springer, Cham (2022). https:\/\/doi.org\/10.1007\/978-3-031-21648-0_30","key":"5_CR22","DOI":"10.1007\/978-3-031-21648-0_30"},{"doi-asserted-by":"crossref","unstructured":"Tian, Z., Shen, C., Chen, H., He, T.: FCOS: fully convolutional one-stage object detection. In: ICCV, pp. 9627\u20139636 (2019)","key":"5_CR23","DOI":"10.1109\/ICCV.2019.00972"},{"doi-asserted-by":"crossref","unstructured":"Cai, Z., Vasconcelos, N.: Cascade R-CNN: delving into high quality object detection. In: CVPR, pp. 6154\u20136162 (2018)","key":"5_CR24","DOI":"10.1109\/CVPR.2018.00644"},{"unstructured":"Redmon, J., Farhadi, A.: Yolov3: an incremental improvement. arXiv preprint arXiv:1804.02767 (2018). 6, 7","key":"5_CR25"},{"key":"5_CR26","first-page":"17721","volume":"33","author":"X Wang","year":"2020","unstructured":"Wang, X., Zhang, R., Kong, T., Li, L., Shen, C.: SOLOv2: dynamic and fast instance segmentation. In NeurIPS 33, 17721\u201317732 (2020)","journal-title":"In NeurIPS"},{"doi-asserted-by":"crossref","unstructured":"Fang, Y., et al.: Instances AsQueries. In: ICCV, pp. 6910\u20136919 (2021)","key":"5_CR27","DOI":"10.1109\/ICCV48922.2021.00683"},{"doi-asserted-by":"crossref","unstructured":"Kong, T.: FoveaBox: beyound anchor-based object detection. IEEE TIP 29, 7389\u20137398 (2020)","key":"5_CR28","DOI":"10.1109\/TIP.2020.3002345"},{"doi-asserted-by":"crossref","unstructured":"Chen, K., et al.: Hybrid task cascade for instance segmentation. In: CVPR, pp. 4974\u20134983 (2019)","key":"5_CR29","DOI":"10.1109\/CVPR.2019.00511"},{"doi-asserted-by":"crossref","unstructured":"Vu, T., Kang, H., Yoo, C.D.: SCNet: training inference sample consistency for instance segmentation. AAAI 35(3), 2701\u20132709 (2021)","key":"5_CR30","DOI":"10.1609\/aaai.v35i3.16374"},{"unstructured":"Zhu, X., Su, W., Lu, L., Li, B., Wang, X., Dai, J.: Deformable DETR: deformable transformers for end-to-end object detection. In: ICLR, pp. 2988\u20132997 (2021)","key":"5_CR31"},{"unstructured":"Hu, J., et al.: ISTR: end-to-end instance segmentation with transformers. arXiv preprint arXiv:2105.00637 (2021). 6, 7","key":"5_CR32"},{"doi-asserted-by":"crossref","unstructured":"Deng, Q., Ibrayim, M., Hamdulla, A., et al.: The YOLO model that still excels in document layout analysis. Signal, Image and Video Processing, pp. 1\u201310 (2023)","key":"5_CR33","DOI":"10.21203\/rs.3.rs-3268193\/v1"}],"container-title":["Lecture Notes in Computer Science","Document Analysis and Recognition - ICDAR 2024"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-70546-5_5","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,10]],"date-time":"2024-09-10T05:04:13Z","timestamp":1725944653000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-70546-5_5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024]]},"ISBN":["9783031705458","9783031705465"],"references-count":33,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-70546-5_5","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2024]]},"assertion":[{"value":"11 September 2024","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ICDAR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Document Analysis and Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Athens","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Greece","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2024","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"30 August 2024","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"4 September 2024","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icdar2024","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icdar2024.net\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}