{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:55:55Z","timestamp":1781535355541,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":83,"publisher":"ACM","license":[{"start":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T00:00:00Z","timestamp":1781481600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc-nd\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2026,6,16]]},"DOI":"10.1145\/3805622.3810794","type":"proceedings-article","created":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:42:57Z","timestamp":1781534577000},"page":"2200-2209","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["CompArt: Operationalizing Aesthetic Alignment in Text-to-Image Generation via Principles of Art"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-9469-7720","authenticated-orcid":false,"given":"Zhe","family":"Jin","sequence":"first","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6097-7807","authenticated-orcid":false,"given":"Tat-Seng","family":"Chua","sequence":"additional","affiliation":[{"name":"School of Computing, National University of Singapore, Singapore, Singapore"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,6,15]]},"reference":[{"key":"e_1_3_3_2_2_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01140"},{"key":"e_1_3_3_2_3_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00537"},{"key":"e_1_3_3_2_4_2","unstructured":"UC Berkeley. [n. d.]. Design Fundamentals: Elements & Principles. https:\/\/guides.lib.berkeley.edu\/c.php?g=920740&p=6634741. (Accessed: 2024-11-12)."},{"key":"e_1_3_3_2_5_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"e_1_3_3_2_6_2","doi-asserted-by":"publisher","unstructured":"Hila Chefer Yuval Alaluf Yael Vinker Lior Wolf and Daniel Cohen-Or. 2023. Attend-and-Excite: Attention-Based Semantic Guidance for Text-to-Image Diffusion Models. ACM Trans. Graph. 42 4 (2023) 148:1\u2013148:10. 10.1145\/3592116","DOI":"10.1145\/3592116"},{"key":"e_1_3_3_2_7_2","first-page":"26561","volume-title":"Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems 2021, NeurIPS 2021, December 6-14, 2021, virtual","author":"Chen Haibo","year":"2021","unstructured":"Haibo Chen, Lei Zhao, Zhizhong Wang, Huiming Zhang, Zhiwen Zuo, Ailin Li, Wei Xing, and Dongming Lu. 2021. Artistic Style Transfer with Internal-external Learning and Contrastive Learning. In Advances in Neural Information Processing Systems 34: Annual Conference on Neural Information Processing Systems 2021, NeurIPS 2021, December 6-14, 2021, virtual. 26561\u201326573."},{"key":"e_1_3_3_2_8_2","doi-asserted-by":"publisher","unstructured":"Xinlei Chen Hao Fang Tsung-Yi Lin Ramakrishna Vedantam Saurabh Gupta Piotr Doll\u00e1r and C.\u00a0Lawrence Zitnick. 2015. Microsoft COCO Captions: Data Collection and Evaluation Server. CoRR abs\/1504.00325 (2015). 10.48550\/ARXIV.1504.00325","DOI":"10.48550\/ARXIV.1504.00325"},{"key":"e_1_3_3_2_9_2","volume-title":"Advances in Neural Information Processing Systems","author":"Cho Jaemin","year":"2023","unstructured":"Jaemin Cho, Abhay Zala, and Mohit Bansal. 2023. Visual Programming for Step-by-Step Text-to-Image Generation and Evaluation. In Advances in Neural Information Processing Systems , Vol.\u00a036."},{"key":"e_1_3_3_2_10_2","doi-asserted-by":"publisher","DOI":"10.1007\/11744078_23"},{"key":"e_1_3_3_2_11_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01104"},{"key":"e_1_3_3_2_12_2","volume-title":"Advances in Neural Information Processing Systems","author":"Dhariwal Prafulla","year":"2021","unstructured":"Prafulla Dhariwal and Alexander\u00a0Quinn Nichol. 2021. Diffusion Models Beat GANs on Image Synthesis. In Advances in Neural Information Processing Systems, A.\u00a0Beygelzimer, Y.\u00a0Dauphin, P.\u00a0Liang, and J.\u00a0Wortman Vaughan (Eds.)."},{"key":"e_1_3_3_2_13_2","volume-title":"Advances in Neural Information Processing Systems","author":"Ding Ming","year":"2021","unstructured":"Ming Ding, Zhuoyi Yang, Wenyi Hong, Wendi Zheng, Chang Zhou, Da Yin, Junyang Lin, Xu Zou, Zhou Shao, Hongxia Yang, and Jie Tang. 2021. CogView: Mastering Text-to-Image Generation via Transformers. In Advances in Neural Information Processing Systems , Vol.\u00a034."},{"key":"e_1_3_3_2_14_2","doi-asserted-by":"publisher","unstructured":"Guian Fang Zutao Jiang Jianhua Han Guangsong Lu Hang Xu and Xiaodan Liang. 2023. Boosting Text-to-Image Diffusion Models with Fine-Grained Semantic Rewards. 10.48550\/ARXIV.2305.19599","DOI":"10.48550\/ARXIV.2305.19599"},{"key":"e_1_3_3_2_15_2","volume-title":"The Eleventh International Conference on Learning Representations","author":"Feng Weixi","year":"2023","unstructured":"Weixi Feng, Xuehai He, Tsu-Jui Fu, Varun Jampani, Arjun\u00a0Reddy Akula, Pradyumna Narayana, Sugato Basu, Xin\u00a0Eric Wang, and William\u00a0Yang Wang. 2023. Training-Free Structured Diffusion Guidance for Compositional Text-to-Image Synthesis. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_2_16_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00454"},{"key":"e_1_3_3_2_17_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-20059-5_41"},{"key":"e_1_3_3_2_18_2","volume-title":"The Eleventh International Conference on Learning Representations","author":"Gal Rinon","year":"2023","unstructured":"Rinon Gal, Yuval Alaluf, Yuval Atzmon, Or Patashnik, Amit\u00a0Haim Bermano, Gal Chechik, and Daniel Cohen-Or. 2023. An Image is Worth One Word: Personalizing Text-to-Image Generation using Textual Inversion. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_2_19_2","doi-asserted-by":"publisher","unstructured":"V\u00edctor Gallego. 2022. Personalizing Text-to-Image Generation via Aesthetic Gradients. CoRR abs\/2209.12330 (2022). 10.48550\/ARXIV.2209.12330","DOI":"10.48550\/ARXIV.2209.12330"},{"key":"e_1_3_3_2_20_2","volume-title":"International Conference on Machine Learning","author":"Ganin Yaroslav","year":"2018","unstructured":"Yaroslav Ganin, Tejas\u00a0D. Kulkarni, Igor Babuschkin, Ali Eslami, and Oriol Vinyals. 2018. Synthesizing Programs for Images using Reinforced Adversarial Learning. In International Conference on Machine Learning."},{"key":"e_1_3_3_2_21_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.265"},{"key":"e_1_3_3_2_22_2","volume-title":"Neural Information Processing Systems","author":"Goodfellow Ian\u00a0J.","year":"2014","unstructured":"Ian\u00a0J. Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, David Warde-Farley, Sherjil Ozair, Aaron\u00a0C. Courville, and Yoshua Bengio. 2014. Generative Adversarial Nets. In Neural Information Processing Systems."},{"key":"e_1_3_3_2_23_2","doi-asserted-by":"publisher","unstructured":"Simone Grassini and Mika Koivisto. 2025. Artificial Creativity? Evaluating AI Against Human Performance in Creative Interpretation of Visual Stimuli. International Journal of Human\u2013Computer Interaction 41 7 (2025) 4037\u20134048. 10.1080\/10447318.2024.2345430","DOI":"10.1080\/10447318.2024.2345430"},{"key":"e_1_3_3_2_24_2","volume-title":"Advances in Neural Information Processing Systems","author":"Gu Yuchao","year":"2023","unstructured":"Yuchao Gu, Xintao Wang, Jay\u00a0Zhangjie Wu, Yujun Shi, Yunpeng Chen, Zihan Fan, Wuyou Xiao, Rui Zhao, Shuning Chang, Weijia Wu, Yixiao Ge, Ying Shan, and Mike\u00a0Zheng Shou. 2023. Mix-of-Show: Decentralized Low-Rank Adaptation for Multi-Concept Customization of Diffusion Models. In Advances in Neural Information Processing Systems , Vol.\u00a036."},{"key":"e_1_3_3_2_25_2","volume-title":"International Conference on Learning Representations","author":"Ha David","year":"2018","unstructured":"David Ha and Douglas Eck. 2018. A Neural Representation of Sketch Drawings. In International Conference on Learning Representations."},{"key":"e_1_3_3_2_26_2","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023","author":"Hao Yaru","year":"2023","unstructured":"Yaru Hao, Zewen Chi, Li Dong, and Furu Wei. 2023. Optimizing Prompts for Text-to-Image Generation. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023, Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (Eds.)."},{"key":"e_1_3_3_2_27_2","volume-title":"The Eleventh International Conference on Learning Representations","author":"Hertz Amir","year":"2023","unstructured":"Amir Hertz, Ron Mokady, Jay Tenenbaum, Kfir Aberman, Yael Pritch, and Daniel Cohen-Or. 2023. Prompt-to-Prompt Image Editing with Cross-Attention Control. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_2_28_2","volume-title":"Proceedings of the 34th International Conference on Neural Information Processing Systems","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. In Proceedings of the 34th International Conference on Neural Information Processing Systems. Article 574."},{"key":"e_1_3_3_2_29_2","doi-asserted-by":"publisher","unstructured":"Jonathan Ho and Tim Salimans. 2022. Classifier-Free Diffusion Guidance. CoRR abs\/2207.12598 (2022). arXiv:https:\/\/arXiv.org\/abs\/2207.1259810.48550\/ARXIV.2207.12598","DOI":"10.48550\/ARXIV.2207.12598"},{"key":"e_1_3_3_2_30_2","volume-title":"International Conference on Learning Representations","author":"Hu Edward\u00a0J.","year":"2022","unstructured":"Edward\u00a0J. Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations."},{"key":"e_1_3_3_2_31_2","doi-asserted-by":"publisher","unstructured":"Xiwei Hu Rui Wang Yixiao Fang Bin Fu Pei Cheng and Gang Yu. 2024. ELLA: Equip Diffusion Models with LLM for Enhanced Semantic Alignment. CoRR abs\/2403.05135 (2024). 10.48550\/ARXIV.2403.05135","DOI":"10.48550\/ARXIV.2403.05135"},{"key":"e_1_3_3_2_32_2","doi-asserted-by":"publisher","DOI":"10.52202\/075280-3443"},{"key":"e_1_3_3_2_33_2","first-page":"13753","volume-title":"Proceedings of the 40th International Conference on Machine Learning","author":"Huang Lianghua","year":"2023","unstructured":"Lianghua Huang, Di Chen, Yu Liu, Yujun Shen, Deli Zhao, and Jingren Zhou. 2023. Composer: Creative and Controllable Image Synthesis with Composable Conditions. In Proceedings of the 40th International Conference on Machine Learning. 13753\u201313773."},{"key":"e_1_3_3_2_34_2","unstructured":"Javier Ja\u00e9n. 2023. Rethink Plastic. Barron\u2019s. https:\/\/www.barrons.com\/articles\/cheap-new-plastic-choking-the-world-9b318936 Accessed: 2026-04-21."},{"key":"e_1_3_3_2_35_2","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/849"},{"key":"e_1_3_3_2_36_2","first-page":"103064","volume-title":"Advances in Neural Information Processing Systems","author":"Jin Xin","year":"2024","unstructured":"Xin Jin, Qianqian Qiao, Yi Lu, Huaye Wang, Heng Huang, Shan Gao, Jianfei Liu, and Rui Li. 2024. APDDv2: Aesthetics of Paintings and Drawings Dataset with Artist Labeled Scores and Comments. In Advances in Neural Information Processing Systems , Vol.\u00a037. 103064\u2013103075."},{"key":"e_1_3_3_2_37_2","volume-title":"3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings","author":"Kingma Diederik\u00a0P.","year":"2015","unstructured":"Diederik\u00a0P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings."},{"key":"e_1_3_3_2_38_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01202"},{"key":"e_1_3_3_2_39_2","doi-asserted-by":"publisher","unstructured":"Pierre Leli\u00e8vre and Peter Neri. 2021. A deep-learning framework for human perception of abstract art composition. Journal of Vision 21 5 (05 2021) 9\u20139. 10.1167\/jov.21.5.9","DOI":"10.1167\/jov.21.5.9"},{"key":"e_1_3_3_2_40_2","doi-asserted-by":"publisher","unstructured":"Jia Li Lei Yao Ella Hendriks and James\u00a0Z. Wang. 2012. Rhythmic Brushstrokes Distinguish van Gogh from His Contemporaries: Findings via Automated Brushstroke Extraction. IEEE Trans. Pattern Anal. Mach. Intell. 34 6 (2012) 1159\u20131176. 10.1109\/TPAMI.2011.203","DOI":"10.1109\/TPAMI.2011.203"},{"key":"e_1_3_3_2_41_2","first-page":"366","volume-title":"34th British Machine Vision Conference 2023, BMVC 2023, Aberdeen, UK, November 20-24, 2023","author":"Li Yumeng","year":"2023","unstructured":"Yumeng Li, Margret Keuper, Dan Zhang, and Anna Khoreva. 2023. Divide & Bind Your Attention for Improved Generative Semantic Nursing. In 34th British Machine Vision Conference 2023, BMVC 2023, Aberdeen, UK, November 20-24, 2023. BMVA Press, 366."},{"key":"e_1_3_3_2_42_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02156"},{"key":"e_1_3_3_2_43_2","unstructured":"Long Lian Boyi Li Adam Yala and Trevor Darrell. 2024. LLM-grounded Diffusion: Enhancing Prompt Understanding of Text-to-Image Diffusion Models with Large Language Models. Trans. Mach. Learn. Res. 2024 (2024)."},{"key":"e_1_3_3_2_44_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v40i21.38811"},{"key":"e_1_3_3_2_45_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19790-1_26"},{"key":"e_1_3_3_2_46_2","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1873965"},{"key":"e_1_3_3_2_47_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.9"},{"key":"e_1_3_3_2_48_2","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-69423-6_2"},{"key":"e_1_3_3_2_49_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02058"},{"key":"e_1_3_3_2_50_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"e_1_3_3_2_51_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2012.6247954"},{"key":"e_1_3_3_2_52_2","doi-asserted-by":"publisher","unstructured":"OpenAI. 2023. GPT-4 Technical Report. CoRR abs\/2303.08774 (2023). 10.48550\/ARXIV.2303.08774","DOI":"10.48550\/ARXIV.2303.08774"},{"key":"e_1_3_3_2_53_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_2_54_2","volume-title":"The Twelfth International Conference on Learning Representations","author":"Podell Dustin","year":"2024","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2024. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. In The Twelfth International Conference on Learning Representations."},{"key":"e_1_3_3_2_55_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612012"},{"key":"e_1_3_3_2_56_2","unstructured":"Colin Raffel Noam Shazeer Adam Roberts Katherine Lee Sharan Narang Michael Matena Yanqi Zhou Wei Li and Peter\u00a0J. Liu. 2020. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. J. Mach. Learn. Res. 21 (2020) 140:1\u2013140:67."},{"key":"e_1_3_3_2_57_2","doi-asserted-by":"publisher","unstructured":"Aditya Ramesh Prafulla Dhariwal Alex Nichol Casey Chu and Mark Chen. 2022. Hierarchical Text-Conditional Image Generation with CLIP Latents. ArXiv (2022). 10.48550\/arXiv.2204.06125","DOI":"10.48550\/arXiv.2204.06125"},{"key":"e_1_3_3_2_58_2","doi-asserted-by":"publisher","unstructured":"Aditya Ramesh Mikhail Pavlov Gabriel Goh Scott Gray Chelsea Voss Alec Radford Mark Chen and Ilya Sutskever. 2021. Zero-Shot Text-to-Image Generation. ArXiv (2021). 10.48550\/arXiv.2102.12092","DOI":"10.48550\/arXiv.2102.12092"},{"key":"e_1_3_3_2_59_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_3_2_60_2","doi-asserted-by":"publisher","unstructured":"Babak Saleh and Ahmed\u00a0M. Elgammal. 2016. Large-scale Classification of Fine-Art Paintings: Learning The Right Metric on The Right Feature. International Journal for Digital Art History2 (2016). 10.11588\/dah.2016.2.23376","DOI":"10.11588\/dah.2016.2.23376"},{"key":"e_1_3_3_2_61_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1238"},{"key":"e_1_3_3_2_62_2","volume-title":"Advances in Neural Information Processing Systems","author":"Sohn Kihyuk","year":"2023","unstructured":"Kihyuk Sohn, Lu Jiang, Jarred Barber, Kimin Lee, Nataniel Ruiz, Dilip Krishnan, Huiwen Chang, Yuanzhen Li, Irfan Essa, Michael Rubinstein, Yuan Hao, Glenn Entis, Irina Blok, and Daniel\u00a0Castro Chin. 2023. StyleDrop: Text-to-Image Synthesis of Any Style. In Advances in Neural Information Processing Systems , Vol.\u00a036."},{"key":"e_1_3_3_2_63_2","volume-title":"International Conference on Learning Representations","author":"Song Jiaming","year":"2021","unstructured":"Jiaming Song, Chenlin Meng, and Stefano Ermon. 2021. Denoising Diffusion Implicit Models. In International Conference on Learning Representations."},{"key":"e_1_3_3_2_64_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV57701.2024.00705"},{"key":"e_1_3_3_2_65_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.naacl-long.304"},{"key":"e_1_3_3_2_66_2","unstructured":"Tarik Takasu. 2016. Van Gogh on a PCB. Behance. https:\/\/www.behance.net\/gallery\/41188523\/Van-Gogh-on-a-PCB Accessed: 2026-04-21."},{"key":"e_1_3_3_2_67_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP.2016.7533051"},{"key":"e_1_3_3_2_68_2","doi-asserted-by":"crossref","unstructured":"Yilin Wang Haiyang Xu Xiang Zhang Zeyuan Chen Zhizhou Sha Zirui Wang and Zhuowen Tu. 2024. OmniControlNet: Dual-stage Integration for Conditional Image Generation. 2024 IEEE\/CVF Conference on Computer Vision and Pattern Recognition Workshops (CVPRW) (2024) 7436\u20137448.","DOI":"10.1109\/CVPRW63382.2024.00739"},{"key":"e_1_3_3_2_69_2","doi-asserted-by":"publisher","DOI":"10.1145\/3591106.3592262"},{"key":"e_1_3_3_2_70_2","volume-title":"Advances in Neural Information Processing Systems","author":"Xu Jiazheng","year":"2023","unstructured":"Jiazheng Xu, Xiao Liu, Yuchen Wu, Yuxuan Tong, Qinkai Li, Ming Ding, Jie Tang, and Yuxiao Dong. 2023. ImageReward: Learning and Evaluating Human Preferences for Text-to-Image Generation. In Advances in Neural Information Processing Systems , Vol.\u00a036."},{"key":"e_1_3_3_2_71_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00972"},{"key":"e_1_3_3_2_72_2","first-page":"56704","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Yang Ling","year":"2024","unstructured":"Ling Yang, Zhaochen Yu, Chenlin Meng, Minkai Xu, Stefano Ermon, and Bin Cui. 2024. Mastering Text-to-Image Diffusion: Recaptioning, Planning, and Generating with Multimodal LLMs. In Proceedings of the 41st International Conference on Machine Learning. 56704\u201356721."},{"key":"e_1_3_3_2_73_2","doi-asserted-by":"publisher","unstructured":"Peter Young Alice Lai Micah Hodosh and Julia Hockenmaier. 2014. From image descriptions to visual denotations: New similarity metrics for semantic inference over event descriptions. Trans. Assoc. Comput. Linguistics 2 (2014) 67\u201378. 10.1162\/TACL_A_00166","DOI":"10.1162\/TACL_A_00166"},{"key":"e_1_3_3_2_74_2","unstructured":"Jiahui Yu Yuanzhong Xu Jing\u00a0Yu Koh Thang Luong Gunjan Baid Zirui Wang Vijay Vasudevan Alexander Ku Yinfei Yang Burcu\u00a0Karagol Ayan Ben Hutchinson Wei Han Zarana Parekh Xin Li Han Zhang Jason Baldridge and Yonghui Wu. 2022. Scaling Autoregressive Models for Content-Rich Text-to-Image Generation. Trans. Mach. Learn. Res. 2022 (2022)."},{"key":"e_1_3_3_2_75_2","volume-title":"The Eleventh International Conference on Learning Representations","author":"Yuksekgonul Mert","year":"2023","unstructured":"Mert Yuksekgonul, Federico Bianchi, Pratyusha Kalluri, Dan Jurafsky, and James Zou. 2023. When and Why Vision-Language Models Behave like Bags-Of-Words, and What to Do About It?. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_3_2_76_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_3_2_77_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00787"},{"key":"e_1_3_3_2_78_2","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i7.28570"},{"key":"e_1_3_3_2_79_2","volume-title":"Advances in Neural Information Processing Systems","author":"Zhao Shihao","year":"2023","unstructured":"Shihao Zhao, Dongdong Chen, Yen-Chun Chen, Jianmin Bao, Shaozhe Hao, Lu Yuan, and Kwan-Yee\u00a0K. Wong. 2023. Uni-ControlNet: All-in-One Control to Text-to-Image Diffusion Models. In Advances in Neural Information Processing Systems , Vol.\u00a036."},{"key":"e_1_3_3_2_80_2","doi-asserted-by":"publisher","DOI":"10.1145\/2647868.2654930"},{"key":"e_1_3_3_2_81_2","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3611863"},{"key":"e_1_3_3_2_82_2","doi-asserted-by":"publisher","unstructured":"Tao Zhou Chen Fang Zhaowen Wang Jimei Yang Byungmoon Kim Zhili Chen Jonathan Brandt and Demetri Terzopoulos. 2018. Learning to Sketch with Deep Q Networks and Demonstrated Strokes. ArXiv (2018). 10.48550\/arXiv.1810.05977","DOI":"10.48550\/arXiv.1810.05977"},{"key":"e_1_3_3_2_83_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.244"},{"key":"e_1_3_3_2_84_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01543"}],"event":{"name":"ICMR '26: International Conference on Multimedia Retrieval","location":"Amsterdam The Netherlands","acronym":"ICMR '26","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 2026 International Conference on Multimedia Retrieval"],"original-title":[],"deposited":{"date-parts":[[2026,6,15]],"date-time":"2026-06-15T14:44:45Z","timestamp":1781534685000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3805622.3810794"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,6,15]]},"references-count":83,"alternative-id":["10.1145\/3805622.3810794","10.1145\/3805622"],"URL":"https:\/\/doi.org\/10.1145\/3805622.3810794","relation":{},"subject":[],"published":{"date-parts":[[2026,6,15]]},"assertion":[{"value":"2026-06-15","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}