{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,16]],"date-time":"2026-01-16T04:21:32Z","timestamp":1768537292664,"version":"3.49.0"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680725","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"10957-10965","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Decoder-Only LLMs are Better Controllers for Diffusion Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-8771-852X","authenticated-orcid":false,"given":"Ziyi","family":"Dong","sequence":"first","affiliation":[{"name":"Sun Yat-sen University, GuangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-2219-7245","authenticated-orcid":false,"given":"Yao","family":"Xiao","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University, GuangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2190-0767","authenticated-orcid":false,"given":"Pengxu","family":"Wei","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University &amp; Pengcheng Laboratory, GuangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2248-3755","authenticated-orcid":false,"given":"Liang","family":"Lin","sequence":"additional","affiliation":[{"name":"Sun Yat-sen University &amp; Pengcheng Laboratory, GuangZhou, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"PromptCrafter: Crafting Text-to-Image Prompt through Mixed-Initiative Dialogue with LLM. CoRR","author":"Baek Seungho","year":"2023","unstructured":"Seungho Baek, Hyerin Im, Jiseung Ryu, Juhyeong Park, and Tak Yeon Lee. 2023. PromptCrafter: Crafting Text-to-Image Prompt through Mixed-Initiative Dialogue with LLM. CoRR, Vol. abs\/2307.08985 (2023). showeprint[arXiv]2307.08985"},{"key":"e_1_3_2_2_2_1","unstructured":"James Betker Gabriel Goh Li Jing TimBrooks Jianfeng Wang Linjie Li LongOuyang JuntangZhuang JoyceLee YufeiGuo WesamManassra PrafullaDhariwal CaseyChu YunxinJiao and Aditya Ramesh. [n. d.]. Improving Image Generation with Better Captions. https:\/\/cdn.openai.com\/papers\/dall-e-3.pdf"},{"key":"e_1_3_2_2_3_1","volume-title":"NeurIPS","author":"Brown Tom B.","year":"2020","unstructured":"Tom B. Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, Sandhini Agarwal, Ariel Herbert-Voss, Gretchen Krueger, Tom Henighan, Rewon Child, Aditya Ramesh, Daniel M. Ziegler, Jeffrey Wu, Clemens Winter, Christopher Hesse, Mark Chen, Eric Sigler, Mateusz Litwin, Scott Gray, Benjamin Chess, Jack Clark, Christopher Berner, Sam McCandlish, Alec Radford, Ilya Sutskever, and Dario Amodei. 2020. Language Models are Few-Shot Learners. In NeurIPS 2020, Hugo Larochelle, Marc'Aurelio Ranzato, Raia Hadsell, Maria-Florina Balcan, and Hsuan-Tien Lin (Eds.)."},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"crossref","unstructured":"Zhihong Chen Guiming Chen Shizhe Diao Xiang Wan and Benyou Wang. 2023. On the Difference of BERT-style and CLIP-style Text Encoders. In ACL.","DOI":"10.18653\/v1\/2023.findings-acl.866"},{"key":"e_1_3_2_2_5_1","volume-title":"RiFeGAN: Rich Feature Generation for Text-to-Image Synthesis From Prior Knowledge. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10908--10917","author":"Cheng Jun","year":"2020","unstructured":"Jun Cheng, Fuxiang Wu, Yanling Tian, Lei Wang, and Dapeng Tao. 2020. RiFeGAN: Rich Feature Generation for Text-to-Image Synthesis From Prior Knowledge. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10908--10917."},{"key":"e_1_3_2_2_6_1","volume-title":"Xing","author":"Chiang Wei-Lin","year":"2023","unstructured":"Wei-Lin Chiang, Zhuohan Li, Zi Lin, Ying Sheng, Zhanghao Wu, Hao Zhang, Lianmin Zheng, Siyuan Zhuang, Yonghao Zhuang, Joseph E. Gonzalez, Ion Stoica, and Eric P. Xing. 2023. Vicuna: An Open-Source Chatbot Impressing GPT-4 with 90%* ChatGPT Quality. https:\/\/lmsys.org\/blog\/2023-03--30-vicuna\/"},{"key":"e_1_3_2_2_7_1","article-title":"PaLM: Scaling Language Modeling with Pathways","volume":"24","author":"Chowdhery Aakanksha","year":"2023","unstructured":"Aakanksha Chowdhery, Sharan Narang, Jacob Devlin, Maarten Bosma, Gaurav Mishra, Adam Roberts, Paul Barham, Hyung Won Chung, Charles Sutton, Sebastian Gehrmann, Parker Schuh, Kensen Shi, Sasha Tsvyashchenko, Joshua Maynez, Abhishek Rao, Parker Barnes, Yi Tay, Noam Shazeer, Vinodkumar Prabhakaran, Emily Reif, Nan Du, Ben Hutchinson, Reiner Pope, James Bradbury, Jacob Austin, Michael Isard, Guy Gur-Ari, Pengcheng Yin, Toju Duke, Anselm Levskaya, Sanjay Ghemawat, Sunipa Dev, Henryk Michalewski, Xavier Garcia, Vedant Misra, Kevin Robinson, Liam Fedus, Denny Zhou, Daphne Ippolito, David Luan, Hyeontaek Lim, Barret Zoph, Alexander Spiridonov, Ryan Sepassi, David Dohan, Shivani Agrawal, Mark Omernick, Andrew M. Dai, Thanumalayan Sankaranarayana Pillai, Marie Pellat, Aitor Lewkowycz, Erica Moreira, Rewon Child, Oleksandr Polozov, Katherine Lee, Zongwei Zhou, Xuezhi Wang, Brennan Saeta, Mark Diaz, Orhan Firat, Michele Catasta, Jason Wei, Kathy Meier-Hellstern, Douglas Eck, Jeff Dean, Slav Petrov, and Noah Fiedel. 2023. PaLM: Scaling Language Modeling with Pathways. J. Mach. Learn. Res., Vol. 24 (2023), 240:1--240:113.","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_2_8_1","unstructured":"et.al. Dan Hendrycks. 2021. Measuring Massive Multitask Language Understanding. In ICLR."},{"key":"e_1_3_2_2_9_1","volume-title":"BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In NAACL-HLT 2019, Jill Burstein, Christy Doran, and Thamar Solorio (Eds.). 4171--4186."},{"key":"e_1_3_2_2_10_1","unstructured":"Ming Ding Zhuoyi Yang Wenyi Hong Wendi Zheng Chang Zhou Da Yin Junyang Lin Xu Zou Zhou Shao Hongxia Yang and Jie Tang. 2021. CogView: Mastering Text-to-Image Generation via Transformers. In Advances in Neural Information Processing Systems. 19822--19835."},{"key":"e_1_3_2_2_11_1","volume-title":"Muzammal Naseer, Salman H. Khan, and Peter Wonka.","author":"Gani Hanan","year":"2023","unstructured":"Hanan Gani, Shariq Farooq Bhat, Muzammal Naseer, Salman H. Khan, and Peter Wonka. 2023. LLM Blueprint: Enabling Text-to-Image Generation with Complex and Detailed Prompts. CoRR, Vol. abs\/2310.10640 (2023). showeprint[arXiv]2310.10640"},{"key":"e_1_3_2_2_12_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising Diffusion Probabilistic Models. In Advances in Neural Information Processing Systems."},{"key":"e_1_3_2_2_13_1","volume-title":"Classifier-Free Diffusion Guidance. In NeurIPS 2021 Workshop on Deep Generative Models and Downstream Applications.","author":"Ho Jonathan","year":"2021","unstructured":"Jonathan Ho and Tim Salimans. 2021. Classifier-Free Diffusion Guidance. In NeurIPS 2021 Workshop on Deep Generative Models and Downstream Applications."},{"key":"e_1_3_2_2_14_1","volume-title":"Llama 2: Open Foundation and Fine-Tuned Chat Models. CoRR","author":"Touvron Hugo","year":"2023","unstructured":"et. al. Hugo Touvron. 2023. Llama 2: Open Foundation and Fine-Tuned Chat Models. CoRR, Vol. abs\/2307.09288 (2023). showeprint[arXiv]2307.09288"},{"key":"e_1_3_2_2_15_1","volume-title":"Zero-Shot Text-Guided Object Generation with Dream Fields. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 857--866","author":"Jain Ajay","year":"2022","unstructured":"Ajay Jain, Ben Mildenhall, Jonathan T. Barron, Pieter Abbeel, and Ben Poole. 2022. Zero-Shot Text-Guided Object Generation with Dream Fields. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 857--866."},{"key":"e_1_3_2_2_16_1","volume-title":"Torr","author":"Li Bowen","year":"2019","unstructured":"Bowen Li, Xiaojuan Qi, Thomas Lukasiewicz, and Philip H. S. Torr. 2019. Controllable Text-to-Image Generation. In NeurIPS 2019, Hanna M. Wallach, Hugo Larochelle, Alina Beygelzimer, Florence d'Alch\u00e9-Buc, Emily B. Fox, and Roman Garnett (Eds.)."},{"key":"e_1_3_2_2_17_1","volume-title":"Suriya Gunasekar, and Yin Tat Lee.","author":"Li Yuanzhi","year":"2023","unstructured":"Yuanzhi Li, S\u00e9bastien Bubeck, Ronen Eldan, Allie Del Giorno, Suriya Gunasekar, and Yin Tat Lee. 2023. Textbooks Are All You Need II: phi-1.5 technical report. CoRR, Vol. abs\/2309.05463 (2023). showeprint[arXiv]2309.05463"},{"key":"e_1_3_2_2_18_1","volume-title":"LLM-grounded Diffusion: Enhancing Prompt Understanding of Text-to-Image Diffusion Models with Large Language Models. CoRR","author":"Lian Long","year":"2023","unstructured":"Long Lian, Boyi Li, Adam Yala, and Trevor Darrell. 2023. LLM-grounded Diffusion: Enhancing Prompt Understanding of Text-to-Image Diffusion Models with Large Language Models. CoRR, Vol. abs\/2305.13655 (2023). showeprint[arXiv]2305.13655"},{"key":"e_1_3_2_2_19_1","volume-title":"Samaneh Azadi, Gong Zhang, Arman Chopikyan, Yuxiao Hu, Humphrey Shi, Anna Rohrbach, and Trevor Darrell.","author":"Liu Xihui","year":"2021","unstructured":"Xihui Liu, Dong Huk Park, Samaneh Azadi, Gong Zhang, Arman Chopikyan, Yuxiao Hu, Humphrey Shi, Anna Rohrbach, and Trevor Darrell. 2021. More Control for Free! Image Synthesis with Semantic Diffusion Guidance. CoRR, Vol. abs\/2112.05744 (2021). showeprint[arXiv]2112.05744"},{"key":"e_1_3_2_2_20_1","volume-title":"RoBERTa: A Robustly Optimized BERT Pretraining Approach. CoRR","author":"Liu Yinhan","year":"2019","unstructured":"Yinhan Liu, Myle Ott, Naman Goyal, Jingfei Du, Mandar Joshi, Danqi Chen, Omer Levy, Mike Lewis, Luke Zettlemoyer, and Veselin Stoyanov. 2019. RoBERTa: A Robustly Optimized BERT Pretraining Approach. CoRR, Vol. abs\/1907.11692 (2019). showeprint[arXiv]1907.11692"},{"key":"e_1_3_2_2_21_1","volume-title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. In International Conference on Machine Learning. 16784--16804","author":"Nichol Alexander Quinn","year":"2022","unstructured":"Alexander Quinn Nichol, Prafulla Dhariwal, Aditya Ramesh, Pranav Shyam, Pamela Mishkin, Bob McGrew, Ilya Sutskever, and Mark Chen. 2022. GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. In International Conference on Machine Learning. 16784--16804."},{"key":"e_1_3_2_2_23_1","volume-title":"Kosmos-2: Grounding Multimodal Large Language Models to the World. ArXiv","author":"Peng Zhiliang","year":"2023","unstructured":"Zhiliang Peng, Wenhui Wang, Li Dong, Yaru Hao, Shaohan Huang, Shuming Ma, and Furu Wei. 2023. Kosmos-2: Grounding Multimodal Large Language Models to the World. ArXiv, Vol. abs\/2306.14824 (2023)."},{"key":"e_1_3_2_2_24_1","volume-title":"SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. CoRR","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2023. SDXL: Improving Latent Diffusion Models for High-Resolution Image Synthesis. CoRR, Vol. abs\/2307.01952 (2023). showeprint[arXiv]2307.01952"},{"key":"e_1_3_2_2_25_1","volume-title":"Proceedings of the International Conference on Machine Learning","volume":"139","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the International Conference on Machine Learning, Vol. 139. 8748--8763."},{"key":"e_1_3_2_2_26_1","article-title":"Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer","volume":"21","author":"Raffel Colin","year":"2020","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J. Liu. 2020. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. J. Mach. Learn. Res., Vol. 21 (2020), 140:1--140:67.","journal-title":"J. Mach. Learn. Res."},{"key":"e_1_3_2_2_27_1","volume-title":"High-Resolution Image Synthesis with Latent Diffusion Models. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10674--10685","author":"Rombach Robin","year":"2022","unstructured":"Robin Rombach, Andreas Blattmann, Dominik Lorenz, Patrick Esser, and Bj\u00f6rn Ommer. 2022. High-Resolution Image Synthesis with Latent Diffusion Models. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 10674--10685."},{"key":"e_1_3_2_2_28_1","volume-title":"Burcu Karagol Ayan, S. Sara Mahdavi, Rapha Gontijo Lopes, Tim Salimans, Jonathan Ho, David J. Fleet, and Mohammad Norouzi.","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily Denton, Seyed Kamyar Seyed Ghasemipour, Burcu Karagol Ayan, S. Sara Mahdavi, Rapha Gontijo Lopes, Tim Salimans, Jonathan Ho, David J. Fleet, and Mohammad Norouzi. 2022. Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding. CoRR, Vol. abs\/2205.11487 (2022). showeprint[arXiv]2205.11487"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v37i2.25353"},{"key":"e_1_3_2_2_30_1","unstructured":"wanng. 2023. midjourney-v5--202304-clean. https:\/\/huggingface.co\/datasets\/wanng\/midjourney-v5--202304-clean."},{"key":"e_1_3_2_2_31_1","volume-title":"Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, Ben Hutchinson, Wei Han, Zarana Parekh, Xin Li, Han Zhang, Jason Baldridge, and Yonghui Wu.","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Yuanzhong Xu, Jing Yu Koh, Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, Ben Hutchinson, Wei Han, Zarana Parekh, Xin Li, Han Zhang, Jason Baldridge, and Yonghui Wu. 2022. Scaling Autoregressive Models for Content-Rich Text-to-Image Generation. CoRR, Vol. abs\/2206.10789 (2022). showeprint[arXiv]2206.10789"},{"key":"e_1_3_2_2_32_1","volume-title":"Sigmoid Loss for Language Image Pre-Training. CoRR","author":"Zhai Xiaohua","year":"2023","unstructured":"Xiaohua Zhai, Basil Mustafa, Alexander Kolesnikov, and Lucas Beyer. 2023. Sigmoid Loss for Language Image Pre-Training. CoRR, Vol. abs\/2303.15343 (2023). showeprint[arXiv]2303.15343"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680725","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680725","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:06:24Z","timestamp":1750291584000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680725"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":31,"alternative-id":["10.1145\/3664647.3680725","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680725","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}