{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T06:19:00Z","timestamp":1778048340931,"version":"3.51.4"},"reference-count":42,"publisher":"IEEE","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026,3,6]]},"DOI":"10.1109\/wacv61042.2026.00092","type":"proceedings-article","created":{"date-parts":[[2026,5,5]],"date-time":"2026-05-05T19:59:32Z","timestamp":1778011172000},"page":"879-887","source":"Crossref","is-referenced-by-count":0,"title":["Analysis of Text Accuracy and Visual Alignment in Vision-Language Models for Artistic Text Generation"],"prefix":"10.1109","author":[{"given":"Fatima","family":"Alderazi","sequence":"first","affiliation":[{"name":"King Fahd University of Petroleum and Minerals,Information and Computer Science Department,Dhahran,Saudi Arabia,31261"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Motaz","family":"Alfarraj","sequence":"additional","affiliation":[{"name":"King Fahd University of Petroleum and Minerals SDAIA-KFUPM Joint Research Center for Artificial Intelligence,Electrical Engineering Department,Dhahran,Saudi Arabia,31261"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","year":"2025","journal-title":"ActiveLoop. Icdar 2013 dataset"},{"key":"ref2","volume-title":"Adobe Systems. Adobe Photoshop User Guide","year":"2021"},{"key":"ref3","article-title":"Are diffusion models vision-and-language reasoners?","year":"2023","journal-title":"NeurIPS 2023"},{"key":"ref4","article-title":"Qwen-vl: A versatile vision-language model for understanding, localization, text reading, and beyond","author":"Bai","year":"2023"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.1986.4767851"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0410"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.52202\/075280-0410"},{"key":"ref8","article-title":"Hybrid gan-transformer frameworks for artistic text synthesis","volume-title":"Proceedings of NeurIPS","author":"Chen"},{"key":"ref9","volume-title":"CorelDRAW: Advanced Graphic Design Tools","year":"2021"},{"key":"ref10","article-title":"An image is worth 16x16 words: Transformers for image recognition at scale","author":"Dosovitskiy","year":"2020"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00716"},{"key":"ref12","volume-title":"Controlnet","year":"2025"},{"key":"ref13","first-page":"117","article-title":"Tokenization","volume-title":"Syntactic word-class tagging","year":"1999"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.254"},{"key":"ref15","article-title":"Refining text-to-image generation using clip-based feedback loops","author":"He","year":"2022"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"ref17","article-title":"Gans trained by a two time-scale update rule converge to a local nash equilibrium","author":"Heusel","year":"2017","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"ref18","year":"2025","journal-title":"ControlNet with Stable Diffusion XL (SDXL) \u2014 Diffusers Documentation"},{"key":"ref19","article-title":"Gan-based stylized text generation for creative applications","volume-title":"Proceedings of the ACM Multimedia Conference","author":"Li"},{"issue":"2","key":"ref20","first-page":"10","article-title":"Clip for perception evaluation in multimodal tasks","volume":"42","author":"Liu","year":"2023","journal-title":"ACM Transactions on Graphics"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-73226-3_21"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3209870"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3209870"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2004-668"},{"key":"ref25","year":"2022","journal-title":"Dall\u00b7e 2"},{"key":"ref26","year":"2023","journal-title":"Gpt-4v(ision) system card"},{"key":"ref27","volume-title":"helpers for computing deltas","year":"2020"},{"key":"ref28","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","volume-title":"International Conference on Machine Learning (ICML)","author":"Radford"},{"issue":"1","key":"ref29","article-title":"Exploring the limits of transfer learning with a unified text-to-text transformer","volume":"21","author":"Raffel","year":"2020","journal-title":"J. Mach. Learn. Res."},{"key":"ref30","article-title":"DALL-E: Creating Images from Text","volume-title":"OpenAI Blog","author":"Ramesh","year":"2021"},{"key":"ref31","article-title":"Exploring Latent Diffusion Models: Stable Diffusion and Beyond","author":"Romero","year":"2022"},{"key":"ref32","article-title":"Imagen: Photorealistic text-to-image diffusion models","author":"Saharia","year":"2022"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.7717\/peerj-cs.3104\/fig-1"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICDAR.2007.4376991"},{"key":"ref35","year":"2024","journal-title":"Qwen2.5: A party of foundation models"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/BigData59044.2023.10386743"},{"key":"ref38","article-title":"Artistic text generation with gans and llms","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR)","author":"Xu"},{"key":"ref39","first-page":"1234","article-title":"Perceptual metrics for evaluating text-to-image synthesis","volume":"30","author":"Zhang","year":"2021","journal-title":"IEEE Transactions on Image Processing"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00068"},{"key":"ref42","article-title":"Hierarchical transformers for artistic text representation in visual contexts","author":"Zhu","year":"2023","journal-title":"IEEE Transactions on Multimedia"}],"event":{"name":"2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)","location":"Tucson, AZ, USA","start":{"date-parts":[[2026,3,6]]},"end":{"date-parts":[[2026,3,10]]}},"container-title":["2026 IEEE\/CVF Winter Conference on Applications of Computer Vision (WACV)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11491838\/11491925\/11492287.pdf?arnumber=11492287","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T05:55:51Z","timestamp":1778046951000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11492287\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":42,"URL":"https:\/\/doi.org\/10.1109\/wacv61042.2026.00092","relation":{},"subject":[],"published":{"date-parts":[[2026,3,6]]}}}