{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,6]],"date-time":"2026-08-06T12:46:23Z","timestamp":1786020383655,"version":"3.56.0"},"reference-count":35,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,1,1]],"date-time":"2027-01-01T00:00:00Z","timestamp":1798761600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,6,30]],"date-time":"2026-06-30T00:00:00Z","timestamp":1782777600000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100021178","name":"University of Nottingham Ningbo China","doi-asserted-by":"publisher","award":["22DF_AGB"],"award-info":[{"award-number":["22DF_AGB"]}],"id":[{"id":"10.13039\/501100021178","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100003847","name":"Ningbo Municipal People&apos;s Government","doi-asserted-by":"publisher","award":["2021B-008-C"],"award-info":[{"award-number":["2021B-008-C"]}],"id":[{"id":"10.13039\/501100003847","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Expert Systems with Applications"],"published-print":{"date-parts":[[2027,1]]},"DOI":"10.1016\/j.eswa.2026.133491","type":"journal-article","created":{"date-parts":[[2026,7,1]],"date-time":"2026-07-01T15:28:13Z","timestamp":1782919693000},"page":"133491","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"PA","title":["High-fidelity text refinement for ControlNet-guided latent diffusion in document inpainting"],"prefix":"10.1016","volume":"332","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0938-0690","authenticated-orcid":false,"given":"Qinglin","family":"Mao","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2191-7187","authenticated-orcid":false,"given":"Shengzhe","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-1225-8151","authenticated-orcid":false,"given":"Songliang","family":"Chen","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Hang","family":"Zhou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0896-0650","authenticated-orcid":false,"given":"Pushpendu","family":"Kar","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6317-5877","authenticated-orcid":false,"given":"Anthony Graham","family":"Bellotti","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.eswa.2026.133491_bib0001","unstructured":"Cicchetti, G., & Comminiello, D. (2024). NAF-DPM: A nonlinear activation-free diffusion probabilistic model for document enhancement. arXiv preprint arXiv: 2404.05669."},{"key":"10.1016\/j.eswa.2026.133491_bib0002","unstructured":"Du, Y., Li, C., Guo, R., Yin, X., Liu, W., Zhou, J., Bai, Y., Yu, Z., Yang, Y., Dang, Q. et al. (2020). PP-OCR: A practical ultra lightweight OCR system. arXiv preprint arXiv: 2009.09941."},{"key":"10.1016\/j.eswa.2026.133491_bib0003","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho","year":"2020","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10.1016\/j.eswa.2026.133491_bib0004","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"4083","article-title":"LayoutLMV3: Pre-training for document AI with unified text and image masking","author":"Huang","year":"2022"},{"key":"10.1016\/j.eswa.2026.133491_bib0005","series-title":"Proceedings of the IEEE conference on computer vision and pattern recognition","first-page":"1125","article-title":"Image-to-image translation with conditional adversarial networks","author":"Isola","year":"2017"},{"key":"10.1016\/j.eswa.2026.133491_bib0006","series-title":"2019 International conference on document analysis and recognition workshops (ICDARW)","first-page":"1","article-title":"FUNSD: A dataset for form understanding in noisy scanned documents","volume":"Vol. 2","author":"Jaume","year":"2019"},{"key":"10.1016\/j.eswa.2026.133491_bib0007","unstructured":"Li, C., Liu, W., Guo, R., Yin, X., Jiang, K., Du, Y., Du, Y., Zhu, L., Lai, B., Hu, X. et al. (2022). PP-OCRV3: More attempts for the improvement of ultra lightweight ocr system. arXiv preprint arXiv: 2206.03001."},{"key":"10.1016\/j.eswa.2026.133491_bib0008","unstructured":"Li, Z., Zhang, J., Lin, Q., Xiong, J., Long, Y., Deng, X., Zhang, Y., Liu, X., Huang, M., Xiao, Z. et al. (2024). Hunyuan-DIT: A powerful multi-resolution diffusion transformer with fine-grained chinese understanding. arXiv preprint arXiv: 2405.08748."},{"key":"10.1016\/j.eswa.2026.133491_bib0009","series-title":"Proceedings of the European conference on computer vision (ECCV)","first-page":"89","article-title":"Image inpainting for irregular holes using partial convolutions","author":"Liu","year":"2018"},{"key":"10.1016\/j.eswa.2026.133491_bib0010","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"4170","article-title":"Coherent semantic attention for image inpainting","author":"Liu","year":"2019"},{"key":"10.1016\/j.eswa.2026.133491_bib0011","doi-asserted-by":"crossref","unstructured":"Lu, W., Su, L., Zheng, J., de Melo, V. V., Shoeleh, F., Hawkin, J. A., Tricco, T., Zhao, H., & Jiang, X. (2025). TextDoctor: Unified document image inpainting via patch pyramid diffusion models. arXiv preprint arXiv: 2503.04021.","DOI":"10.2139\/ssrn.5108776"},{"key":"10.1016\/j.eswa.2026.133491_bib0012","series-title":"Proceedings of the 32nd ACM international conference on multimedia","first-page":"7085","article-title":"HandRefiner: Refining malformed hands in generated images by diffusion-based conditional inpainting","author":"Lu","year":"2024"},{"key":"10.1016\/j.eswa.2026.133491_bib0013","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"11461","article-title":"Repaint: Inpainting using denoising diffusion probabilistic models","author":"Lugmayr","year":"2022"},{"key":"10.1016\/j.eswa.2026.133491_bib0014","doi-asserted-by":"crossref","DOI":"10.1016\/j.neucom.2024.128897","article-title":"ALDII: Adaptive learning-based document image inpainting to enhance the handwritten chinese character legibility of human and machine","volume":"616","author":"Mao","year":"2025","journal-title":"Neurocomputing"},{"key":"10.1016\/j.eswa.2026.133491_sbref0015","first-page":"57","article-title":"Image database TID2013: Peculiarities, results and perspectives","volume":"30","author":"Ponomarenko","year":"2015","journal-title":"Signal Processing: Image Communication"},{"key":"10.1016\/j.eswa.2026.133491_bib0016","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"12977","article-title":"Learning to localize objects improves spatial reasoning in visual-LLMs","author":"Ranasinghe","year":"2024"},{"key":"10.1016\/j.eswa.2026.133491_bib0017","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"10684","article-title":"High-resolution image synthesis with latent diffusion models","author":"Rombach","year":"2022"},{"issue":"11","key":"10.1016\/j.eswa.2026.133491_bib0018","doi-asserted-by":"crossref","first-page":"2298","DOI":"10.1109\/TPAMI.2016.2646371","article-title":"An end-to-end trainable neural network for image-based sequence recognition and its application to scene text recognition","volume":"39","author":"Shi","year":"2016","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"10.1016\/j.eswa.2026.133491_bib0019","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"1147","article-title":"CharFormer: A glyph fusion based attentive framework for high-precision character image denoising","author":"Shi","year":"2022"},{"key":"10.1016\/j.eswa.2026.133491_bib0020","series-title":"Proceedings of the 30th ACM international conference on multimedia","first-page":"1177","article-title":"RCRN: Real-world character image restoration network via skeleton extraction","author":"Shi","year":"2022"},{"key":"10.1016\/j.eswa.2026.133491_bib0021","doi-asserted-by":"crossref","first-page":"5166","DOI":"10.1109\/TMM.2022.3189245","article-title":"TSINIT: A two-stage inpainting network for incomplete text","volume":"25","author":"Sun","year":"2022","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133491_bib0022","unstructured":"Wang, P., Bai, S., Tan, S., Wang, S., Fan, Z., Bai, J., Chen, K., Liu, X., Wang, J., Ge, W. et al. (2024). Qwen2-VL: Enhancing vision-language model\u2019s perception of the world at any resolution. arXiv preprint arXiv: 2409.12191."},{"key":"10.1016\/j.eswa.2026.133491_bib0023","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"17683","article-title":"UFormer: A general u-shaped transformer for image restoration","author":"Wang","year":"2022"},{"key":"10.1016\/j.eswa.2026.133491_bib0024","doi-asserted-by":"crossref","first-page":"3343","DOI":"10.1109\/TMM.2025.3535316","article-title":"All-in-one weather-degraded image restoration via adaptive degradation-aware self-prompting model","volume":"27","author":"Wen","year":"2025","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133491_bib0025","doi-asserted-by":"crossref","first-page":"3380","DOI":"10.1109\/TMM.2026.3651132","article-title":"Structure-preserving frequency-regularized text-guided optimal transport for unpaired rain streaks and raindrops removal","volume":"28","author":"Wen","year":"2026","journal-title":"IEEE Transactions on Multimedia"},{"key":"10.1016\/j.eswa.2026.133491_bib0026","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1109\/TNNLS.2026.3673760","article-title":"When optimal transport meets photo-realistic image dehazing with unpaired training","author":"Wen","year":"2026","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"key":"10.1016\/j.eswa.2026.133491_bib0027","unstructured":"World Wide Web Consortium(2003). Portable network graphics (PNG) specification (second edition). W3C Recommendation and ISO\/IEC 15948:2003https:\/\/www.w3.org\/TR\/2003\/REC-PNG-20031110\/."},{"key":"10.1016\/j.eswa.2026.133491_bib0028","series-title":"Proceedings of the 26th ACM SIGKDD international conference on knowledge discovery & data mining","first-page":"1192","article-title":"LayoutLM: Pre-training of text and layout for document image understanding","author":"Xu","year":"2020"},{"key":"10.1016\/j.eswa.2026.133491_bib0029","series-title":"Proceedings of the 59th annual meeting of the association for computational linguistics and the 11th international joint conference on natural language processing (volume 1: Long papers)","first-page":"2579","article-title":"LayoutLMV2: Multi-modal pre-training for visually-rich document understanding","author":"Xu","year":"2021"},{"key":"10.1016\/j.eswa.2026.133491_bib0030","series-title":"Proceedings of the 31st ACM international conference on multimedia","first-page":"2795","article-title":"DOCDiff: Document enhancement via residual diffusion models","author":"Yang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133491_bib0031","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"5728","article-title":"Restormer: Efficient transformer for high-resolution image restoration","author":"Zamir","year":"2022"},{"key":"10.1016\/j.eswa.2026.133491_bib0032","series-title":"Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition","first-page":"15654","article-title":"DocRes: A generalist model toward unifying document image restoration tasks","author":"Zhang","year":"2024"},{"key":"10.1016\/j.eswa.2026.133491_bib0033","series-title":"Proceedings of the IEEE\/CVF international conference on computer vision","first-page":"3836","article-title":"Adding conditional control to text-to-image diffusion models","author":"Zhang","year":"2023"},{"key":"10.1016\/j.eswa.2026.133491_bib0034","doi-asserted-by":"crossref","DOI":"10.1016\/j.eswa.2024.126224","article-title":"Context-aware mutual learning for blind image inpainting and beyond","volume":"268","author":"Zhao","year":"2025","journal-title":"Expert Systems with Applications"},{"key":"10.1016\/j.eswa.2026.133491_bib0035","series-title":"Proceedings of the AAAI conference on artificial intelligence","first-page":"7775","article-title":"Text image inpainting via global structure-guided diffusion models","volume":"Vol. 38","author":"Zhu","year":"2024"}],"container-title":["Expert Systems with Applications"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426024000?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0957417426024000?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,8,6]],"date-time":"2026-08-06T11:47:03Z","timestamp":1786016823000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0957417426024000"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,1]]},"references-count":35,"alternative-id":["S0957417426024000"],"URL":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133491","relation":{},"ISSN":["0957-4174"],"issn-type":[{"value":"0957-4174","type":"print"}],"subject":[],"published":{"date-parts":[[2027,1]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"High-fidelity text refinement for ControlNet-guided latent diffusion in document inpainting","name":"articletitle","label":"Article Title"},{"value":"Expert Systems with Applications","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.eswa.2026.133491","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Author(s). Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"133491"}}