{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:06:35Z","timestamp":1765343195956,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":35,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755613","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:27:39Z","timestamp":1761377259000},"page":"10379-10387","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["RealText: Realistic Text Image Generation based on Glyph and Scene Aware Inpainting"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0000-0529-0228","authenticated-orcid":false,"given":"Zihou","family":"Liu","sequence":"first","affiliation":[{"name":"School of Information Science and Technology, Beijing University of Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1237-7177","authenticated-orcid":false,"given":"Dongming","family":"Zhang","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Communication Content Cognition, People's Daily Online, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1290-0738","authenticated-orcid":false,"given":"Jing","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, Beijing University of Technology, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-3176-8030","authenticated-orcid":false,"given":"Jun","family":"Li","sequence":"additional","affiliation":[{"name":"State Key Laboratory of Communication Content Cognition, People's Daily Online, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-0066-3448","authenticated-orcid":false,"given":"Yongdong","family":"Zhang","sequence":"additional","affiliation":[{"name":"School of Information Science and Technology, University of Science and Technology of China, Hefei, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Jaided AI. 2020. EasyOCR. https:\/\/github.com\/JaidedAI\/EasyOCR\/."},{"key":"e_1_3_2_1_2_1","volume-title":"Kandinsky 3: Text-to-Image Synthesis for Multifunctional Generative Framework. CoRR","author":"Arkhipkin Vladimir","year":"2024","unstructured":"Vladimir Arkhipkin, Viacheslav Vasilev, Andrei Filatov, Igor Pavlov, Julia Agafonova, Nikolai Gerasimenko, Anna Averchenkova, Evelina Mironova, Anton Bukashkin, Konstantin Kulikov, Andrey Kuznetsov, and Denis Dimitrov. 2024. Kandinsky 3: Text-to-Image Synthesis for Multifunctional Generative Framework. CoRR, Vol. abs\/2410.21061 (2024)."},{"key":"e_1_3_2_1_3_1","unstructured":"BlackForestLab. 2024. Flux.1. https:\/\/blackforestlabs.ai\/."},{"key":"e_1_3_2_1_4_1","volume-title":"Conference on Neural Information Processing Systems","author":"Chen Jingye","year":"2023","unstructured":"Jingye Chen, Yupan Huang, Tengchao Lv, Lei Cui, Qifeng Chen, and Furu Wei. 2023. TextDiffuser: Diffusion Models as Text Painters. In Conference on Neural Information Processing Systems. New Orleans, LA, USA, December 10 - 16, 2023."},{"key":"e_1_3_2_1_5_1","volume-title":"European Conference on Computer Vision","author":"Chen Jingye","year":"2024","unstructured":"Jingye Chen, Yupan Huang, Tengchao Lv, Lei Cui, Qifeng Chen, and Furu Wei. 2024. TextDiffuser-2: Unleashing the Power of Language Models for Text Rendering. In European Conference on Computer Vision. Milan, Italy, September 29-October 4, 2024."},{"key":"e_1_3_2_1_6_1","unstructured":"DeepFloyd. 2023. DeepFloyd IF. https:\/\/github.com\/deep-floyd\/IF\/."},{"key":"e_1_3_2_1_7_1","volume-title":"Scaling Rectified Flow Transformers for High-Resolution Image Synthesis. In International Conference on Machine Learning","author":"Esser Patrick","year":"2024","unstructured":"Patrick Esser, Sumith Kulal, Andreas Blattmann, Rahim Entezari, Jonas M\u00fcller, Harry Saini, Yam Levi, Dominik Lorenz, Axel Sauer, Frederic Boesel, Dustin Podell, Tim Dockhorn, Zion English, and Robin Rombach. 2024. Scaling Rectified Flow Transformers for High-Resolution Image Synthesis. In International Conference on Machine Learning. Vienna, Austria, July 21-27, 2024."},{"key":"e_1_3_2_1_8_1","unstructured":"Blender Foundation. 1994. Blender. https:\/\/github.com\/blender\/."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1145\/3422622"},{"key":"e_1_3_2_1_10_1","volume-title":"Synthetic Data for Text Localisation in Natural Images. In Conference on Computer Vision and Pattern Recognition","author":"Gupta Ankush","year":"2016","unstructured":"Ankush Gupta, Andrea Vedaldi, and Andrew Zisserman. 2016. Synthetic Data for Text Localisation in Natural Images. In Conference on Computer Vision and Pattern Recognition. Las Vegas, NV, USA, June 27-30, 2016."},{"key":"e_1_3_2_1_11_1","volume-title":"CLIPScore: A Reference-free Evaluation Metric for Image Captioning. In Conference on Empirical Methods in Natural Language Processing. Virtual \/ Punta Cana","author":"Hessel Jack","year":"2021","unstructured":"Jack Hessel, Ari Holtzman, Maxwell Forbes, Ronan Le Bras, and Yejin Choi. 2021. CLIPScore: A Reference-free Evaluation Metric for Image Captioning. In Conference on Empirical Methods in Natural Language Processing. Virtual \/ Punta Cana, Dominican Republic, November 7-11, 2021."},{"key":"e_1_3_2_1_12_1","volume-title":"Denoising Diffusion Probabilistic Models. In Conference on Neural Information Processing Systems. virtual","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. In Conference on Neural Information Processing Systems. virtual, December 6-12, 2020."},{"key":"e_1_3_2_1_13_1","volume-title":"Repurposing Diffusion-Based Image Generators for Monocular Depth Estimation. In Conference on Computer Vision and Pattern Recognition","author":"Ke Bingxin","year":"2024","unstructured":"Bingxin Ke, Anton Obukhov, Shengyu Huang, Nando Metzger, Rodrigo Caye Daudt, and Konrad Schindler. 2024. Repurposing Diffusion-Based Image Generators for Monocular Depth Estimation. In Conference on Computer Vision and Pattern Recognition. Seattle, WA, USA, June 16-22, 2024."},{"key":"e_1_3_2_1_14_1","volume-title":"PP-OCRv3: More Attempts for the Improvement of Ultra Lightweight OCR System. CoRR","author":"Li Chenxia","year":"2022","unstructured":"Chenxia Li, Weiwei Liu, Ruoyu Guo, Xiaoting Yin, Kaitao Jiang, Yongkun Du, Yuning Du, Lingfeng Zhu, Baohua Lai, Xiaoguang Hu, Dianhai Yu, and Yanjun Ma. 2022. PP-OCRv3: More Attempts for the Improvement of Ultra Lightweight OCR System. CoRR, Vol. abs\/2206.03001 (2022)."},{"key":"e_1_3_2_1_15_1","volume-title":"GlyphDraw: Learning to Draw Chinese Characters in Image Synthesis Models Coherently. CoRR","author":"Ma Jian","year":"2023","unstructured":"Jian Ma, Mingjun Zhao, Chen Chen, Ruichen Wang, Di Niu, Haonan Lu, and Xiaodong Lin. 2023. GlyphDraw: Learning to Draw Chinese Characters in Image Synthesis Models Coherently. CoRR, Vol. abs\/2303.17870 (2023)."},{"key":"e_1_3_2_1_16_1","volume-title":"Improved Denoising Diffusion Probabilistic Models. In International Conference on Machine Learning. Virtual","author":"Nichol Alexander Quinn","year":"2021","unstructured":"Alexander Quinn Nichol and Prafulla Dhariwal. 2021. Improved Denoising Diffusion Probabilistic Models. In International Conference on Machine Learning. Virtual, July 18-24, 2021."},{"key":"e_1_3_2_1_17_1","volume-title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. In International Conference on Machine Learning","author":"Nichol Alexander Quinn","year":"2022","unstructured":"Alexander Quinn Nichol, Prafulla Dhariwal, Aditya Ramesh, Pranav Shyam, Pamela Mishkin, Bob McGrew, Ilya Sutskever, and Mark Chen. 2022. GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. In International Conference on Machine Learning. Baltimore, Maryland, USA, July 17-23, 2022."},{"key":"e_1_3_2_1_18_1","unstructured":"OpenAI. 2023. DALL\u00b7E3. https:\/\/openai.com\/index\/dall-e-3\/."},{"key":"e_1_3_2_1_19_1","volume-title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. In International Conference on Learning Representations","author":"Podell Dustin","year":"2024","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2024. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. In International Conference on Learning Representations. Vienna, Austria, May 7-11, 2024."},{"key":"e_1_3_2_1_20_1","volume-title":"Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. Virtual","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In International Conference on Machine Learning. Virtual, July 18-24, 2021."},{"key":"e_1_3_2_1_21_1","article-title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J. Liu. 2020. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. Journal of Machine Learning Research, Vol. 21 (2020), 140:1-140:67.","journal-title":"Journal of Machine Learning Research"},{"key":"e_1_3_2_1_22_1","volume-title":"Hierarchical Text-Conditional Image Generation with CLIP Latents. CoRR","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical Text-Conditional Image Generation with CLIP Latents. CoRR, Vol. abs\/2204.06125 (2022)."},{"key":"e_1_3_2_1_23_1","volume-title":"International Conference on Machine Learning","author":"Reed Scott E.","year":"2016","unstructured":"Scott E. Reed, Zeynep Akata, Xinchen Yan, Lajanugen Logeswaran, Bernt Schiele, and Honglak Lee. 2016. Generative Adversarial Text to Image Synthesis. In International Conference on Machine Learning. New York City, NY, USA, June 19-24, 2016."},{"key":"e_1_3_2_1_24_1","volume-title":"High-Resolution Image Synthesis with Latent Diffusion Models. In Computer Vision and Pattern Recognition Conference","author":"Rombach Robin","year":"2022","unstructured":"Robin Rombach, Andreas Blattmann, Dominik Lorenz, Patrick Esser, and Bj\u00f6rn Ommer. 2022. High-Resolution Image Synthesis with Latent Diffusion Models. In Computer Vision and Pattern Recognition Conference. New Orleans, LA, USA, June 18-24, 2022."},{"key":"e_1_3_2_1_25_1","volume-title":"Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding. In Conference on Neural Information Processing Systems","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily L. Denton, Seyed Kamyar Seyed Ghasemipour, Raphael Gontijo Lopes, Burcu Karagol Ayan, Tim Salimans, Jonathan Ho, David J. Fleet, and Mohammad Norouzi. 2022. Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding. In Conference on Neural Information Processing Systems. New Orleans, LA, USA, November 28 - December 9, 2022."},{"key":"e_1_3_2_1_26_1","volume-title":"Denoising Diffusion Implicit Models. In International Conference on Learning Representations. Virtual","author":"Song Jiaming","year":"2021","unstructured":"Jiaming Song, Chenlin Meng, and Stefano Ermon. 2021. Denoising Diffusion Implicit Models. In International Conference on Learning Representations. Virtual, May 3-7, 2021."},{"key":"e_1_3_2_1_27_1","volume-title":"AnyText: Multilingual Visual Text Generation and Editing. In International Conference on Learning Representations","author":"Tuo Yuxiang","year":"2024","unstructured":"Yuxiang Tuo, Wangmeng Xiang, Jun-Yan He, Yifeng Geng, and Xuansong Xie. 2024. AnyText: Multilingual Visual Text Generation and Editing. In International Conference on Learning Representations. Vienna, Austria, May 7-11, 2024."},{"key":"e_1_3_2_1_28_1","volume-title":"Conference on Neural Information Processing Systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N. Gomez, Lukasz Kaiser, and Illia Polosukhin. 2017. Attention is All you Need. In Conference on Neural Information Processing Systems. Long Beach, CA, USA, December 4-9, 2017."},{"key":"e_1_3_2_1_29_1","volume-title":"Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis. CoRR","author":"Wu Xiaoshi","year":"2023","unstructured":"Xiaoshi Wu, Yiming Hao, Keqiang Sun, Yixiong Chen, Feng Zhu, Rui Zhao, and Hongsheng Li. 2023. Human Preference Score v2: A Solid Benchmark for Evaluating Human Preferences of Text-to-Image Synthesis. CoRR, Vol. abs\/2306.09341 (2023)."},{"key":"e_1_3_2_1_30_1","volume-title":"GlyphControl: Glyph Conditional Control for Visual Text Generation. In Conference on Neural Information Processing Systems","author":"Yang Yukang","year":"2023","unstructured":"Yukang Yang, Dongnan Gui, Yuhui Yuan, Weicong Liang, Haisong Ding, Han Hu, and Kai Chen. 2023. GlyphControl: Glyph Conditional Control for Visual Text Generation. In Conference on Neural Information Processing Systems. New Orleans, LA, USA, December 10 - 16, 2023."},{"key":"e_1_3_2_1_31_1","volume-title":"Verisimilar Image Synthesis for Accurate Detection and Recognition of Texts in Scenes. In European Conference on Computer Vision","author":"Zhan Fangneng","year":"2018","unstructured":"Fangneng Zhan, Shijian Lu, and Chuhui Xue. 2018. Verisimilar Image Synthesis for Accurate Detection and Recognition of Texts in Scenes. In European Conference on Computer Vision. Munich, Germany, September 8-14, 2018."},{"key":"e_1_3_2_1_32_1","volume-title":"Spatial Fusion GAN for Image Synthesis. In Conference on Computer Vision and Pattern Recognition","author":"Zhan Fangneng","year":"2019","unstructured":"Fangneng Zhan, Hongyuan Zhu, and Shijian Lu. 2019. Spatial Fusion GAN for Image Synthesis. In Conference on Computer Vision and Pattern Recognition. Long Beach, CA, USA, June 16-20, 2019."},{"key":"e_1_3_2_1_33_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i7.28550"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_1_35_1","volume-title":"European Conference on Computer Vision","author":"Zhao Yiming","year":"2024","unstructured":"Yiming Zhao and Zhouhui Lian. 2024. UDiffText: A Unified Framework for High-Quality Text Synthesis in Arbitrary Images via Character-Aware Diffusion Models. In European Conference on Computer Vision. Milan, Italy, September 29-October 4, 2024."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755613","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:03:29Z","timestamp":1765343009000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755613"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":35,"alternative-id":["10.1145\/3746027.3755613","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755613","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}