{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,22]],"date-time":"2026-08-22T13:25:34Z","timestamp":1787405134876,"version":"build-2736575974"},"publisher-location":"Cham","reference-count":29,"publisher":"Springer Nature Switzerland","isbn-type":[{"value":"9783032360328","type":"print"},{"value":"9783032360335","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T00:00:00Z","timestamp":1787443200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,8,23]],"date-time":"2026-08-23T00:00:00Z","timestamp":1787443200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2027]]},"DOI":"10.1007\/978-3-032-36033-5_11","type":"book-chapter","created":{"date-parts":[[2026,8,22]],"date-time":"2026-08-22T13:21:55Z","timestamp":1787404915000},"page":"177-194","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Doc2Doc: Structure-Aware Generative Rendering for\u00a0Bi-directional Document Translation"],"prefix":"10.1007","author":[{"given":"Fahad","family":"Alotaibi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Daulet","family":"Toibazar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Renad","family":"Almusaad","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ranya","family":"Alkahtani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haneen","family":"Alhomoud","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Asma","family":"Ibrahim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yazeed","family":"Alharbi","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Murtadha","family":"Aljubran","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Pedro","family":"Moreno","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2026,8,23]]},"reference":[{"key":"11_CR1","doi-asserted-by":"publisher","first-page":"3358","DOI":"10.1109\/TPAMI.2025.3530998","volume":"47","author":"Z Zhang","year":"2025","unstructured":"Zhang, Z., Zhang, Y., Liang, Y., et al.: Understand layout and translate text: unified feature-conductive end-to-end document image translation. IEEE Trans. Pattern Anal. Mach. Intell. 47, 3358\u20133376 (2025)","journal-title":"IEEE Trans. Pattern Anal. Mach. Intell."},{"key":"11_CR2","doi-asserted-by":"crossref","unstructured":"Zhang, P., Li, C., Qiao, L.: VSR: a unified framework for document layout analysis combining vision, semantics and relations. arXiv preprint arXiv:2105.06220 (2021)","DOI":"10.1007\/978-3-030-86549-8_8"},{"key":"11_CR3","unstructured":"Zhao, Z., Kang, H., Wang, B., He, C.: DocLayout-YOLO: enhancing document layout analysis through diverse synthetic data and global-to-local adaptive perception. arXiv preprint arXiv:2410.12628 (2024)"},{"key":"11_CR4","doi-asserted-by":"crossref","unstructured":"Xu, Y., Xu, Y., Lv, T., et al.: LayoutLMv2: multi-modal pre-training for visually-rich document understanding. In: Proceedings of the 59th Annual Meeting of the Association for Computational Linguistics (ACL), pp. 2579\u20132591 (2021)","DOI":"10.18653\/v1\/2021.acl-long.201"},{"key":"11_CR5","doi-asserted-by":"crossref","unstructured":"Huang, Y., Lv, T., Cui, L.: LayoutLMv3: pre-training for document AI with unified text and image masking. In: Proceedings of the 30th ACM International Conference on Multimedia, pp. 4083\u20134091 (2022)","DOI":"10.1145\/3503161.3548112"},{"key":"11_CR6","doi-asserted-by":"crossref","unstructured":"Wang, D., Raman, N., Sibue, M.: DocLLM: A layout-aware generative language model for multimodal document understanding. arXiv preprint arXiv:2401.00908 (2023)","DOI":"10.18653\/v1\/2024.acl-long.463"},{"key":"11_CR7","doi-asserted-by":"crossref","unstructured":"Wang, J., Jin, L., Ding, K.: LiLT: a simple yet effective language-independent layout transformer for structured document understanding. arXiv preprint arXiv:2202.13669 (2022)","DOI":"10.18653\/v1\/2022.acl-long.534"},{"key":"11_CR8","doi-asserted-by":"crossref","unstructured":"Shehzadi, T., Stricker, D., Afzal, M.: A hybrid approach for document layout analysis in document images. arXiv preprint arXiv:2404.17888 (2024)","DOI":"10.1007\/978-3-031-70546-5_2"},{"key":"11_CR9","doi-asserted-by":"publisher","first-page":"317","DOI":"10.1007\/s10032-024-00473-y","volume":"27","author":"S Biswas","year":"2024","unstructured":"Biswas, S., Llad\u00f3s, J., Pal, U.: SemiDocSeg: harnessing semi-supervised learning for document layout analysis. Int. J. Document Anal. Recogn. (IJDAR) 27, 317\u2013334 (2024)","journal-title":"Int. J. Document Anal. Recogn. (IJDAR)"},{"key":"11_CR10","doi-asserted-by":"crossref","unstructured":"Xu, Y., Li, M., Cui, L., et al.: LayoutLM: pre-training of text and layout for document image understanding. In: Proceedings of the 26th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining (KDD), pp. 1192\u20131200 (2020)","DOI":"10.1145\/3394486.3403172"},{"key":"11_CR11","doi-asserted-by":"crossref","unstructured":"Zhang, N., Cheng, H., Chen, J.: M2Doc: a multi-modal fusion approach for document layout analysis. In: AAAI Conference on Artificial Intelligence, pp. 7233\u20137241 (2024)","DOI":"10.1609\/aaai.v38i7.28552"},{"key":"11_CR12","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2022.109046","volume":"134","author":"H Xiang","year":"2022","unstructured":"Xiang, H., Zou, Q., Nawaz, M., et al.: Deep learning for image inpainting: a survey. Pattern Recogn. 134, 109046 (2022)","journal-title":"Pattern Recogn."},{"key":"11_CR13","unstructured":"Du, Y., Liu, C., Chen, D.: PP-OCR: A Practical Ultra Lightweight OCR System. arXiv preprint arXiv:2009.09941 (2020)"},{"key":"11_CR14","doi-asserted-by":"crossref","unstructured":"Suvorov, R., Logacheva, E., Mashikhin, A.: Resolution-robust large mask inpainting with Fourier convolutions. In: Proceedings of the IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV), pp. 2149\u20132159 (2022)","DOI":"10.1109\/WACV51458.2022.00323"},{"key":"11_CR15","unstructured":"Finkelstein, M., Caswell, I., Domhan, T., et al.: TranslateGemma Technical report. arXiv preprint arXiv:2601.09012 (2026)"},{"key":"11_CR16","doi-asserted-by":"crossref","unstructured":"Zhang, Z., et al.: LayoutDIT: layout-aware end-to-end document image translation with multi-step conductive decoder. In: Findings of the Association for Computational Linguistics: EMNLP, pp. 10043\u201310053 (2023)","DOI":"10.18653\/v1\/2023.findings-emnlp.673"},{"key":"11_CR17","doi-asserted-by":"crossref","unstructured":"Liang, Y., et al.: Document image machine translation with dynamic multi-pre-trained models assembling. In: NAACL (Long Papers), pp. 7084\u20137095 (2024)","DOI":"10.18653\/v1\/2024.naacl-long.392"},{"key":"11_CR18","doi-asserted-by":"crossref","unstructured":"Liang, Y., et al.: Single-to-mix modality alignment with multimodal large language model for document image machine translation. In: ACL (Long Papers), pp. 12391\u201312408 (2025)","DOI":"10.18653\/v1\/2025.acl-long.606"},{"key":"11_CR19","doi-asserted-by":"crossref","unstructured":"Zhang, Y., et al.: ICDAR 2025 Competition on End-to-End Document Image Machine Translation Towards Complex Layouts. arXiv preprint arXiv:2603.09392 (2026)","DOI":"10.1007\/978-3-032-04630-7_29"},{"key":"11_CR20","doi-asserted-by":"crossref","unstructured":"Wang, Z., Xu, Y., Cui, L., Shang, J., Wei, F.: LayoutReader: pre-training of text and layout for reading order detection. In: EMNLP, 4735\u20134744 (2021)","DOI":"10.18653\/v1\/2021.emnlp-main.389"},{"key":"11_CR21","doi-asserted-by":"crossref","unstructured":"Zhang, C., et al.: Modeling layout reading order as ordering relations for visually-rich document understanding. In: EMNLP, pp. 9658\u20139678 (2024)","DOI":"10.18653\/v1\/2024.emnlp-main.540"},{"key":"11_CR22","doi-asserted-by":"crossref","unstructured":"Bukhari, S.S., Shafait, F., Breuel, T.M.: High performance layout analysis of Arabic and Urdu document images. In: International Conference on Document Analysis and Recognition (ICDAR), pp. 1275\u20131279 (2011)","DOI":"10.1109\/ICDAR.2011.257"},{"key":"11_CR23","unstructured":"The Unicode Consortium: Unicode Bidirectional Algorithm (UBA) \/ Bidirectional Text FAQ. https:\/\/www.unicode.org\/faq\/bidi.html. Accessed 15 June 2026"},{"key":"11_CR24","unstructured":"Kulshreshtha, P.: Feature Refinement to Improve High Resolution Image Inpainting. arXiv preprint arXiv:2206.13644 (2022)"},{"key":"11_CR25","doi-asserted-by":"crossref","unstructured":"Tanveer, N., Ul-Hasan, A., Shafait, F.: Diffusion models for document image generation. In: ICDAR (LNCS), pp. 438\u2013453 (2023)","DOI":"10.1007\/978-3-031-41682-8_27"},{"key":"11_CR26","doi-asserted-by":"crossref","unstructured":"He, L., Lu, Y., Corring, J., Florencio, D., Zhang, C.: Diffusion-based document layout generation. In: Document Analysis and Recognition - ICDAR (LNCS), pp. 361\u2013378 (2023)","DOI":"10.1007\/978-3-031-41676-7_21"},{"key":"11_CR27","doi-asserted-by":"crossref","unstructured":"Al-Homoud, H., et al.: Cross-Lingual SynthDocs: A Large-Scale Synthetic Corpus for Any to Arabic OCR and Document Understanding. arXiv preprint arXiv:2511.04699 (2025)","DOI":"10.1109\/ICoDSE68111.2025.11351903"},{"key":"11_CR28","unstructured":"Livathinos, N., Auer, C., Nassar, A.: Advanced Layout Analysis Models for Docling. arXiv preprint arXiv:2509.11720 (2025)"},{"key":"11_CR29","unstructured":"Qwen Team: Qwen2.5-VL Technical report. arXiv preprint arXiv:2502.13923 (2025)"}],"container-title":["Lecture Notes in Computer Science","Document Analysis and Recognition \u2013 ICDAR 2026"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-032-36033-5_11","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,8,22]],"date-time":"2026-08-22T13:21:57Z","timestamp":1787404917000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-032-36033-5_11"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,8,23]]},"ISBN":["9783032360328","9783032360335"],"references-count":29,"URL":"https:\/\/doi.org\/10.1007\/978-3-032-36033-5_11","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,8,23]]},"assertion":[{"value":"23 August 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this article.","order":1,"name":"Ethics","label":"Disclosure of Interests","group":{"name":"EthicsHeading","label":"Ethics"}},{"value":"ICDAR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Document Analysis and Recognition","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Vienna","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Austria","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31 August 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2 September 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"20","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"icdar2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/icdar2026.org\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}