{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,13]],"date-time":"2026-06-13T07:21:08Z","timestamp":1781335268116,"version":"3.54.1"},"publisher-location":"New York, NY, USA","reference-count":37,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681495","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"10716-10724","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":8,"title":["Prompt2Poster: Automatically Artistic Chinese Poster Creation from Prompt Only"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-4968-0492","authenticated-orcid":false,"given":"Shaodong","family":"Wang","sequence":"first","affiliation":[{"name":"School of Electronic and Computer Engineering, Peking University &amp; Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9525-9079","authenticated-orcid":false,"given":"Yunyang","family":"Ge","sequence":"additional","affiliation":[{"name":"School of Electronic and Computer Engineering, Peking University &amp; Rabbitpre Intelligence, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6083-8529","authenticated-orcid":false,"given":"Liuhan","family":"Chen","sequence":"additional","affiliation":[{"name":"School of Electronic and Computer Engineering, Peking University &amp; Rabbitpre Intelligence, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-3616-9120","authenticated-orcid":false,"given":"Haiyang","family":"Zhou","sequence":"additional","affiliation":[{"name":"School of Electronic and Computer Engineering, Peking University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3354-3998","authenticated-orcid":false,"given":"Qian","family":"Wang","sequence":"additional","affiliation":[{"name":"School of Electronic and Computer Engineering, Peking University &amp; Rabbitpre Intelligence, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9034-279X","authenticated-orcid":false,"given":"Xinhua","family":"Cheng","sequence":"additional","affiliation":[{"name":"School of Electronic and Computer Engineering, Peking University &amp; Rabbitpre Intelligence, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2120-5588","authenticated-orcid":false,"given":"Li","family":"Yuan","sequence":"additional","affiliation":[{"name":"School of Electronic and Computer Engineering, Peking University &amp; Pengcheng Laboratory, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"2023. DALL-E 3. https:\/\/openai.com\/dall-e-3."},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01343"},{"key":"e_1_3_2_1_3_1","unstructured":"Yogesh Balaji Seungjun Nah Xun Huang Arash Vahdat Jiaming Song Karsten Kreis Miika Aittala Timo Aila Samuli Laine Bryan Catanzaro et al. 2022. ediffi: Text-to-image diffusion models with an ensemble of expert denoisers. arXiv preprint arXiv:2211.01324 (2022)."},{"key":"e_1_3_2_1_4_1","first-page":"1877","article-title":"Language models are few-shot learners","volume":"33","author":"Brown Tom","year":"2020","unstructured":"Tom Brown, Benjamin Mann, Nick Ryder, Melanie Subbiah, Jared D Kaplan, Prafulla Dhariwal, Arvind Neelakantan, Pranav Shyam, Girish Sastry, Amanda Askell, et al. 2020. Language models are few-shot learners. Advances in Neural Information Processing Systems (NeurIPS) 33 (2020), 1877--1901.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548332"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1145\/3172944.3173001"},{"key":"e_1_3_2_1_7_1","volume-title":"Xin Eric Wang, and William Yang Wang","author":"Feng Weixi","year":"2023","unstructured":"Weixi Feng, Wanrong Zhu, Tsu-jui Fu, Varun Jampani, Arjun Akula, Xuehai He, Sugato Basu, Xin Eric Wang, and William Yang Wang. 2023. LayoutGPT: Compositional Visual Planning and Generation with Large Language Models. arXiv preprint arXiv:2305.15393 (2023)."},{"key":"e_1_3_2_1_8_1","volume-title":"TextPainter: Multimodal Text Image Generation withVisual-harmony and Text-comprehension for Poster Design. arXiv preprint arXiv:2308.04733","author":"Gao Yifan","year":"2023","unstructured":"Yifan Gao, Jinpeng Lin, Min Zhou, Chuanbin Liu, Hongtao Xie, Tiezheng Ge, and Yuning Jiang. 2023. TextPainter: Multimodal Text Image Generation withVisual-harmony and Text-comprehension for Poster Design. arXiv preprint arXiv:2308.04733 (2023)."},{"key":"e_1_3_2_1_9_1","volume-title":"Generative adversarial nets. Advances in Neural Information Processing Systems (NeurIPS) 27","author":"Goodfellow Ian","year":"2014","unstructured":"Ian Goodfellow, Jean Pouget-Abadie, Mehdi Mirza, Bing Xu, David Warde-Farley, Sherjil Ozair, Aaron Courville, and Yoshua Bengio. 2014. Generative adversarial nets. Advances in Neural Information Processing Systems (NeurIPS) 27 (2014)."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3411764.3445117"},{"key":"e_1_3_2_1_11_1","volume-title":"Diff-Font: Diffusion Model for Robust One-Shot Font Generation. arXiv preprint arXiv:2212.05895","author":"He Haibin","year":"2022","unstructured":"Haibin He, Xinyuan Chen, Chaoyue Wang, Juhua Liu, Bo Du, Dacheng Tao, and Yu Qiao. 2022. Diff-Font: Diffusion Model for Robust One-Shot Font Generation. arXiv preprint arXiv:2212.05895 (2022)."},{"key":"e_1_3_2_1_12_1","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in Neural Information Processing Systems (NeurIPS) 33 (2020), 6840--6851.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00583"},{"key":"e_1_3_2_1_14_1","volume-title":"Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685","author":"Hu Edward J","year":"2021","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:2106.09685 (2021)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/2449396.2449411"},{"key":"e_1_3_2_1_16_1","volume-title":"Theory of brightness and color contrast in human vision. Vision research 4, 1--2","author":"Jameson Dorothea","year":"1964","unstructured":"Dorothea Jameson and Leo M Hurvich. 1964. Theory of brightness and color contrast in human vision. Vision research 4, 1--2 (1964), 135--154."},{"key":"e_1_3_2_1_17_1","volume-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 7482--7491","author":"Kendall Alex","year":"2018","unstructured":"Alex Kendall, Yarin Gal, and Roberto Cipolla. 2018. Multi-task learning using uncertainty to weigh losses for scene geometry and semantics. In Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition (CVPR). 7482--7491."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TVCG.2020.2999335"},{"key":"e_1_3_2_1_19_1","volume-title":"LLM-grounded Diffusion: Enhancing Prompt Understanding of Text-to-Image Diffusion Models with Large Language Models. arXiv preprint arXiv:2305.13655","author":"Lian Long","year":"2023","unstructured":"Long Lian, Boyi Li, Adam Yala, and Trevor Darrell. 2023. LLM-grounded Diffusion: Enhancing Prompt Understanding of Text-to-Image Diffusion Models with Large Language Models. arXiv preprint arXiv:2305.13655 (2023)."},{"key":"e_1_3_2_1_20_1","volume-title":"LLMgrounded Video Diffusion Models. arXiv preprint arXiv:2309.17444","author":"Lian Long","year":"2023","unstructured":"Long Lian, Baifeng Shi, Adam Yala, Trevor Darrell, and Boyi Li. 2023. LLMgrounded Video Diffusion Models. arXiv preprint arXiv:2309.17444 (2023)."},{"key":"e_1_3_2_1_21_1","volume-title":"Auxiliary tasks in multi-task learning. arXiv preprint arXiv:1805.06334","author":"Liebel Lukas","year":"2018","unstructured":"Lukas Liebel and Marco K\u00f6rner. 2018. Auxiliary tasks in multi-task learning. arXiv preprint arXiv:1805.06334 (2018)."},{"key":"e_1_3_2_1_22_1","volume-title":"AutoPoster: A Highly Automatic and Content-aware Design System for Advertising Poster Generation. arXiv preprint arXiv:2308.01095","author":"Lin Jinpeng","year":"2023","unstructured":"Jinpeng Lin, Min Zhou, Ye Ma, Yifan Gao, Chenxi Fei, Yangjian Chen, Zhang Yu, and Tiezheng Ge. 2023. AutoPoster: A Highly Automatic and Content-aware Design System for Advertising Poster Generation. arXiv preprint arXiv:2308.01095 (2023)."},{"key":"e_1_3_2_1_24_1","volume-title":"Grounded Text-to-Image Synthesis with Attention Refocusing. arXiv preprint arXiv:2306.05427","author":"Phung Quynh","year":"2023","unstructured":"Quynh Phung, Songwei Ge, and Jia-Bin Huang. 2023. Grounded Text-to-Image Synthesis with Attention Refocusing. arXiv preprint arXiv:2306.05427 (2023)."},{"key":"e_1_3_2_1_25_1","volume-title":"Sdxl: improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2023. Sdxl: improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952 (2023)."},{"key":"e_1_3_2_1_26_1","volume-title":"Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 1, 2","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 1, 2 (2022), 3."},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_28_1","first-page":"36479","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume":"35","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily L Denton, Kamyar Ghasemipour, Raphael Gontijo Lopes, Burcu Karagol Ayan, Tim Salimans, et al. 2022. Photorealistic text-to-image diffusion models with deep language understanding. Advances in Neural Information Processing Systems (NeurIPS) 35 (2022), 36479--36494.","journal-title":"Advances in Neural Information Processing Systems (NeurIPS)"},{"key":"e_1_3_2_1_29_1","unstructured":"Mohammad Amin Shabani Zhaowen Wang Difan Liu Nanxuan Zhao Jimei Yang and Yasutaka Furukawa. [n. d.]. Visual Layout Composer: Image-Vector Dual Diffusion Model for Design Layout Generation. ([n. d.])."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/FUZZ.2002.1005020"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491101.3519610"},{"key":"e_1_3_2_1_32_1","volume-title":"Fashion Recommender Systems","author":"Vempati Sreekanth","unstructured":"Sreekanth Vempati, Korah T Malayil, V Sruthi, and R Sandeep. 2020. Enabling hyper-personalisation: Automated ad creative generation and ranking for fashion e-commerce. In Fashion Recommender Systems. Springer, 25--48."},{"key":"e_1_3_2_1_33_1","volume-title":"Desigen: A Pipeline for Controllable Design Template Generation. arXiv preprint arXiv:2403.09093","author":"Weng Haohan","year":"2024","unstructured":"Haohan Weng, Danqing Huang, Yu Qiao, Zheng Hu, Chin-Yew Lin, Tong Zhang, and CL Chen. 2024. Desigen: A Pipeline for Controllable Design Template Generation. arXiv preprint arXiv:2403.09093 (2024)."},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00509"},{"key":"e_1_3_2_1_35_1","volume-title":"Chinese CLIP: Contrastive Vision-Language Pretraining in Chinese. arXiv preprint arXiv:2211.01335","author":"Yang An","year":"2022","unstructured":"An Yang, Junshu Pan, Junyang Lin, Rui Men, Yichang Zhang, Jingren Zhou, and Chang Zhou. 2022. Chinese CLIP: Contrastive Vision-Language Pretraining in Chinese. arXiv preprint arXiv:2211.01335 (2022)."},{"key":"e_1_3_2_1_36_1","volume-title":"GlyphControl: Glyph Conditional Control for Visual Text Generation. arXiv preprint arXiv:2305.18259","author":"Yang Yukang","year":"2023","unstructured":"Yukang Yang, Dongnan Gui, Yuhui Yuan, Haisong Ding, Han Hu, and Kai Chen. 2023. GlyphControl: Glyph Conditional Control for Visual Text Generation. arXiv preprint arXiv:2305.18259 (2023)."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"crossref","unstructured":"Lvmin Zhang Anyi Rao and Maneesh Agrawala. 2023. Adding Conditional Control to Text-to-Image Diffusion Models.","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_2_1_38_1","volume-title":"Composition-aware graphic layout GAN for visual-textual presentation designs. arXiv preprint arXiv:2205.00303","author":"Zhou Min","year":"2022","unstructured":"Min Zhou, Chenchen Xu, Ye Ma, Tiezheng Ge, Yuning Jiang, and Weiwei Xu. 2022. Composition-aware graphic layout GAN for visual-textual presentation designs. arXiv preprint arXiv:2205.00303 (2022)."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681495","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681495","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T00:57:48Z","timestamp":1750294668000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681495"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":37,"alternative-id":["10.1145\/3664647.3681495","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681495","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}