{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,6]],"date-time":"2025-12-06T17:19:53Z","timestamp":1765041593298,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":51,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62325206, 61936005"],"award-info":[{"award-number":["62325206, 61936005"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Postgraduate Research & Practice Innovation Program of Jiangsu Province","award":["KYCX22_0947"],"award-info":[{"award-number":["KYCX22_0947"]}]},{"name":"Key Research and Development Program of Jiangsu Province","award":["BE2023016-4"],"award-info":[{"award-number":["BE2023016-4"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3680873","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:27Z","timestamp":1729925967000},"page":"10659-10668","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["CoIn: A Lightweight and Effective Framework for Story Visualization and Continuation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-4662-7170","authenticated-orcid":false,"given":"Ming","family":"Tao","sequence":"first","affiliation":[{"name":"Nanjing University of Posts and Telecommunications &amp; Peng Cheng Laboratory, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5956-831X","authenticated-orcid":false,"given":"Bing-Kun","family":"Bao","sequence":"additional","affiliation":[{"name":"Nanjing University of Posts and Telecommunications &amp; Peng Cheng Laboratory, Nanjing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2077-1246","authenticated-orcid":false,"given":"Hao","family":"Tang","sequence":"additional","affiliation":[{"name":"Peking University, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2197-9038","authenticated-orcid":false,"given":"Yaowei","family":"Wang","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8343-9665","authenticated-orcid":false,"given":"Changsheng","family":"Xu","sequence":"additional","affiliation":[{"name":"Peng Cheng Laboratory &amp; Institute of Automation, Chinese Academy of Sciences, Shenzhen, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","unstructured":"Yogesh Balaji Seungjun Nah Xun Huang Arash Vahdat Jiaming Song Qinsheng Zhang Karsten Kreis Miika Aittala Timo Aila Samuli Laine et al. 2022. ediff-i: Text-to-image diffusion models with an ensemble of expert denoisers. arXiv preprint arXiv:2211.01324 (2022)."},{"key":"e_1_3_2_1_2_1","volume-title":"Character-Centric Story Visualization via Visual Planning and Token Alignment. arXiv preprint arXiv:2210.08465","author":"Chen Hong","year":"2022","unstructured":"Hong Chen, Rujun Han, Te-Lin Wu, Hideki Nakayama, and Nanyun Peng. 2022. Character-Centric Story Visualization via Visual Planning and Token Alignment. arXiv preprint arXiv:2210.08465 (2022)."},{"key":"e_1_3_2_1_3_1","volume-title":"Diffusion models beat gans on image synthesis. NeurIPS","author":"Dhariwal Prafulla","year":"2021","unstructured":"Prafulla Dhariwal and Alexander Nichol. 2021. Diffusion models beat gans on image synthesis. NeurIPS (2021)."},{"key":"e_1_3_2_1_4_1","volume-title":"Cogview: Mastering text-to-image generation via transformers. In NeurIPS.","author":"Ding Ming","year":"2021","unstructured":"Ming Ding, Zhuoyi Yang, Wenyi Hong, Wendi Zheng, Chang Zhou, Da Yin, Junyang Lin, Xu Zou, Zhou Shao, Hongxia Yang, et al. 2021. Cogview: Mastering text-to-image generation via transformers. In NeurIPS."},{"key":"e_1_3_2_1_5_1","volume-title":"CogView2: Faster and Better Text-to-Image Generation via Hierarchical Transformers. arXiv preprint arXiv:2204.14217","author":"Ding Ming","year":"2022","unstructured":"Ming Ding, Wendi Zheng, Wenyi Hong, and Jie Tang. 2022. CogView2: Faster and Better Text-to-Image Generation via Hierarchical Transformers. arXiv preprint arXiv:2204.14217 (2022)."},{"key":"e_1_3_2_1_6_1","volume-title":"Make-a-scene: Scene-based text-to-image generation with human priors. In ECCV.","author":"Gafni Oran","year":"2022","unstructured":"Oran Gafni, Adam Polyak, Oron Ashual, Shelly Sheynin, Devi Parikh, and Yaniv Taigman. 2022. Make-a-scene: Scene-based text-to-image generation with human priors. In ECCV."},{"key":"e_1_3_2_1_7_1","volume-title":"TaleCrafter: Interactive Story Visualization with Multiple Characters. arXiv preprint arXiv:2305.18247","author":"Gong Yuan","year":"2023","unstructured":"Yuan Gong, Youxin Pang, Xiaodong Cun, Menghan Xia, Haoxin Chen, Longyue Wang, Yong Zhang, Xintao Wang, Ying Shan, and Yujiu Yang. 2023. TaleCrafter: Interactive Story Visualization with Multiple Characters. arXiv preprint arXiv:2305.18247 (2023)."},{"key":"e_1_3_2_1_8_1","unstructured":"Ian Goodfellow Jean Pouget-Abadie Mehdi Mirza Bing Xu David Warde-Farley Sherjil Ozair Aaron Courville and Yoshua Bengio. 2014. Generative adversarial nets. In NeurIPS."},{"key":"e_1_3_2_1_9_1","unstructured":"Shuyang Gu Dong Chen Jianmin Bao Fang Wen Bo Zhang Dongdong Chen Lu Yuan and Baining Guo. 2022. Vector quantized diffusion model for text-to-image synthesis. In CVPR."},{"key":"e_1_3_2_1_10_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01237-3_37"},{"key":"e_1_3_2_1_11_1","unstructured":"Martin Heusel Hubert Ramsauer Thomas Unterthiner Bernhard Nessler and Sepp Hochreiter. 2017. Gans trained by a two time-scale update rule converge to a local nash equilibrium. In NeurIPS."},{"key":"e_1_3_2_1_12_1","volume-title":"Denoising diffusion probabilistic models. NeurIPS","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. NeurIPS (2020)."},{"key":"e_1_3_2_1_13_1","first-page":"47","article-title":"Cascaded Diffusion Models for High Fidelity Image Generation","volume":"23","author":"Ho Jonathan","year":"2022","unstructured":"Jonathan Ho, Chitwan Saharia, William Chan, David J Fleet, Mohammad Norouzi, and Tim Salimans. 2022. Cascaded Diffusion Models for High Fidelity Image Generation. JMLR, Vol. 23 (2022), 47--1.","journal-title":"JMLR"},{"key":"e_1_3_2_1_14_1","volume-title":"Deepstory: Video story qa by deep embedded memory networks. arXiv preprint arXiv:1707.00836","author":"Kim Kyung-Min","year":"2017","unstructured":"Kyung-Min Kim, Min-Oh Heo, Seong-Ho Choi, and Byoung-Tak Zhang. 2017. Deepstory: Video story qa by deep embedded memory networks. arXiv preprint arXiv:1707.00836 (2017)."},{"key":"e_1_3_2_1_15_1","volume-title":"Adam: A method for stochastic optimization. In ICLR.","author":"Kingma Diederik P","year":"2015","unstructured":"Diederik P Kingma and Jimmy Ba. 2015. Adam: A method for stochastic optimization. In ICLR."},{"key":"e_1_3_2_1_16_1","volume-title":"Word-Level Fine-Grained Story Visualization. In European Conference on Computer Vision. Springer, 347--362","author":"Li Bowen","year":"2022","unstructured":"Bowen Li. 2022. Word-Level Fine-Grained Story Visualization. In European Conference on Computer Vision. Springer, 347--362."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548034"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00649"},{"key":"e_1_3_2_1_19_1","volume-title":"Intelligent Grimm--Open-ended Visual Storytelling via Latent Diffusion Models. arXiv preprint arXiv:2306.00973","author":"Liu Chang","year":"2023","unstructured":"Chang Liu, Haoning Wu, Yujie Zhong, Xiaoyun Zhang, and Weidi Xie. 2023. Intelligent Grimm--Open-ended Visual Storytelling via Latent Diffusion Models. arXiv preprint arXiv:2306.00973 (2023)."},{"key":"e_1_3_2_1_20_1","volume-title":"Linguistic and Commonsense Structure into Story Visualization. arXiv preprint arXiv:2110.10834","author":"Maharana Adyasha","year":"2021","unstructured":"Adyasha Maharana and Mohit Bansal. 2021. Integrating Visuospatial, Linguistic and Commonsense Structure into Story Visualization. arXiv preprint arXiv:2110.10834 (2021)."},{"key":"e_1_3_2_1_21_1","volume-title":"Improving generation and evaluation of visual stories via semantic consistency. arXiv preprint arXiv:2105.10026","author":"Maharana Adyasha","year":"2021","unstructured":"Adyasha Maharana, Darryl Hannan, and Mohit Bansal. 2021. Improving generation and evaluation of visual stories via semantic consistency. arXiv preprint arXiv:2105.10026 (2021)."},{"key":"e_1_3_2_1_22_1","volume-title":"StoryDALL-E: Adapting Pretrained Text-to-Image Transformers for Story Continuation. arXiv preprint arXiv:2209.06192","author":"Maharana Adyasha","year":"2022","unstructured":"Adyasha Maharana, Darryl Hannan, and Mohit Bansal. 2022. StoryDALL-E: Adapting Pretrained Text-to-Image Transformers for Story Continuation. arXiv preprint arXiv:2209.06192 (2022)."},{"key":"e_1_3_2_1_23_1","unstructured":"Alexander Quinn Nichol and Prafulla Dhariwal. 2021. Improved denoising diffusion probabilistic models. In ICML."},{"key":"e_1_3_2_1_24_1","volume-title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. In ICML.","author":"Nichol Alexander Quinn","year":"2022","unstructured":"Alexander Quinn Nichol, Prafulla Dhariwal, Aditya Ramesh, Pranav Shyam, Pamela Mishkin, Bob Mcgrew, Ilya Sutskever, and Mark Chen. 2022. GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. In ICML."},{"key":"e_1_3_2_1_25_1","unstructured":"Maxime Oquab Timoth\u00e9e Darcet Th\u00e9o Moutakanni Huy Vo Marc Szafraniec Vasil Khalidov Pierre Fernandez Daniel Haziza Francisco Massa Alaaeldin El-Nouby et al. 2023. Dinov2: Learning robust visual features without supervision. arXiv preprint arXiv:2304.07193 (2023)."},{"key":"e_1_3_2_1_26_1","volume-title":"Synthesizing Coherent Story with Auto-Regressive Latent Diffusion Models. arXiv preprint arXiv:2211.10950","author":"Pan Xichen","year":"2022","unstructured":"Xichen Pan, Pengda Qin, Yuhong Li, Hui Xue, and Wenhu Chen. 2022. Synthesizing Coherent Story with Auto-Regressive Latent Diffusion Models. arXiv preprint arXiv:2211.10950 (2022)."},{"key":"e_1_3_2_1_27_1","volume-title":"Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952","author":"Podell Dustin","year":"2023","unstructured":"Dustin Podell, Zion English, Kyle Lacey, Andreas Blattmann, Tim Dockhorn, Jonas M\u00fcller, Joe Penna, and Robin Rombach. 2023. Sdxl: Improving latent diffusion models for high-resolution image synthesis. arXiv preprint arXiv:2307.01952 (2023)."},{"key":"e_1_3_2_1_28_1","volume-title":"International Conference on Machine Learning. PMLR, 8748--8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al. 2021. Learning transferable visual models from natural language supervision. In International Conference on Machine Learning. PMLR, 8748--8763."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00246"},{"key":"e_1_3_2_1_30_1","volume-title":"Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125","author":"Ramesh Aditya","year":"2022","unstructured":"Aditya Ramesh, Prafulla Dhariwal, Alex Nichol, Casey Chu, and Mark Chen. 2022. Hierarchical text-conditional image generation with clip latents. arXiv preprint arXiv:2204.06125 (2022)."},{"key":"e_1_3_2_1_31_1","unstructured":"Aditya Ramesh Mikhail Pavlov Gabriel Goh Scott Gray Chelsea Voss Alec Radford Mark Chen and Ilya Sutskever. 2021. Zero-shot text-to-image generation. In ICML."},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"crossref","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2022. High-resolution image synthesis with latent diffusion models. In CVPR.","DOI":"10.1109\/CVPR52688.2022.01042"},{"key":"e_1_3_2_1_33_1","unstructured":"Robin Rombach and Patrick Esser. [n. d.]. Stable diffusion mboxv1--4. https:\/\/huggingface.co\/CompVis\/stable-diffusion-v1--4."},{"key":"e_1_3_2_1_34_1","volume-title":"Sanghun Cho and Woonhyuk Baek","author":"Doyup Lee Saehoon Kim Chiheon Kim","year":"2021","unstructured":"Chiheon Kim Doyup Lee Saehoon Kim, Sanghun Cho and Woonhyuk Baek. 2021. minDALL-E on Conceptual Captions. https:\/\/github.com\/kakaobrain\/minDALL-E."},{"key":"e_1_3_2_1_35_1","volume-title":"Burcu Karagol Ayan, S Sara Mahdavi, Rapha Gontijo Lopes, et al.","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily Denton, Seyed Kamyar Seyed Ghasemipour, Burcu Karagol Ayan, S Sara Mahdavi, Rapha Gontijo Lopes, et al. 2022. Photorealistic Text-to-Image Diffusion Models with Deep Language Understanding. arXiv preprint arXiv:2205.11487 (2022)."},{"key":"e_1_3_2_1_36_1","volume-title":"International conference on machine learning. PMLR, 30105--30118","author":"Sauer Axel","year":"2023","unstructured":"Axel Sauer, Tero Karras, Samuli Laine, Andreas Geiger, and Timo Aila. 2023. Stylegan-t: Unlocking the power of gans for fast large-scale text-to-image synthesis. In International conference on machine learning. PMLR, 30105--30118."},{"key":"e_1_3_2_1_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3650033"},{"key":"e_1_3_2_1_38_1","unstructured":"Jascha Sohl-Dickstein Eric Weiss Niru Maheswaranathan and Surya Ganguli. 2015. Deep unsupervised learning using nonequilibrium thermodynamics. In ICML."},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings, Part XVII 16","author":"Song Yun-Zhu","year":"2020","unstructured":"Yun-Zhu Song, Zhi Rui Tam, Hung-Jen Chen, Huiao-Han Lu, and Hong-Han Shuai. 2020. Character-preserving coherent story visualization. In Computer Vision--ECCV 2020: 16th European Conference, Glasgow, UK, August 23--28, 2020, Proceedings, Part XVII 16. Springer, 18--33."},{"key":"e_1_3_2_1_40_1","volume-title":"StoryImager: A Unified and Efficient Framework for Coherent Story Visualization and Completion. arXiv preprint arXiv:2404.05979","author":"Tao Ming","year":"2024","unstructured":"Ming Tao, Bing-Kun Bao, Hao Tang, Yaowei Wang, and Changsheng Xu. 2024. StoryImager: A Unified and Efficient Framework for Coherent Story Visualization and Completion. arXiv preprint arXiv:2404.05979 (2024)."},{"key":"e_1_3_2_1_41_1","volume-title":"GALIP: Generative Adversarial CLIPs for Text-to-Image Synthesis. arXiv preprint arXiv:2301.12959","author":"Tao Ming","year":"2023","unstructured":"Ming Tao, Bing-Kun Bao, Hao Tang, and Changsheng Xu. 2023. GALIP: Generative Adversarial CLIPs for Text-to-Image Synthesis. arXiv preprint arXiv:2301.12959 (2023)."},{"key":"e_1_3_2_1_42_1","volume-title":"Df-gan: A Simple and Effective Baseline for Text-to-Image Synthesis. In CVPR.","author":"Tao Ming","year":"2022","unstructured":"Ming Tao, Hao Tang, Fei Wu, Xiao-Yuan Jing, Bing-Kun Bao, and Xu Changsheng. 2022. Df-gan: A Simple and Effective Baseline for Text-to-Image Synthesis. In CVPR."},{"key":"e_1_3_2_1_43_1","volume-title":"Visual Autoregressive Modeling: Scalable Image Generation via Next-Scale Prediction. arXiv preprint arXiv:2404.02905","author":"Tian Keyu","year":"2024","unstructured":"Keyu Tian, Yi Jiang, Zehuan Yuan, Bingyue Peng, and Liwei Wang. 2024. Visual Autoregressive Modeling: Scalable Image Generation via Next-Scale Prediction. arXiv preprint arXiv:2404.02905 (2024)."},{"key":"e_1_3_2_1_44_1","volume-title":"Attngan: Fine-grained text to image generation with attentional generative adversarial networks. In CVPR.","author":"Xu Tao","year":"2018","unstructured":"Tao Xu, Pengchuan Zhang, Qiuyuan Huang, Han Zhang, Zhe Gan, Xiaolei Huang, and Xiaodong He. 2018. Attngan: Fine-grained text to image generation with attentional generative adversarial networks. In CVPR."},{"key":"e_1_3_2_1_45_1","volume-title":"Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, et al.","author":"Yu Jiahui","year":"2022","unstructured":"Jiahui Yu, Yuanzhong Xu, Jing Yu Koh, Thang Luong, Gunjan Baid, Zirui Wang, Vijay Vasudevan, Alexander Ku, Yinfei Yang, Burcu Karagol Ayan, et al. 2022. Scaling autoregressive models for content-rich text-to-image generation. arXiv preprint arXiv:2206.10789 (2022)."},{"key":"e_1_3_2_1_46_1","unstructured":"Han Zhang Ian Goodfellow Dimitris Metaxas and Augustus Odena. 2019. Self-attention generative adversarial networks. In ICML."},{"key":"e_1_3_2_1_47_1","volume-title":"Stackgan: Text to photo-realistic image synthesis with stacked generative adversarial networks. In ICCV.","author":"Zhang Han","year":"2017","unstructured":"Han Zhang, Tao Xu, Hongsheng Li, Shaoting Zhang, Xiaogang Wang, Xiaolei Huang, and Dimitris N Metaxas. 2017. Stackgan: Text to photo-realistic image synthesis with stacked generative adversarial networks. In ICCV."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2856256"},{"key":"e_1_3_2_1_49_1","volume-title":"CogView3: Finer and Faster Text-to-Image Generation via Relay Diffusion. arXiv preprint arXiv:2403.05121","author":"Zheng Wendi","year":"2024","unstructured":"Wendi Zheng, Jiayan Teng, Zhuoyi Yang, Weihan Wang, Jidong Chen, Xiaotao Gu, Yuxiao Dong, Ming Ding, and Jie Tang. 2024. CogView3: Finer and Faster Text-to-Image Generation via Relay Diffusion. arXiv preprint arXiv:2403.05121 (2024)."},{"key":"e_1_3_2_1_50_1","volume-title":"Dm-gan: Dynamic memory generative adversarial networks for text-to-image synthesis. In CVPR.","author":"Zhu Minfeng","year":"2019","unstructured":"Minfeng Zhu, Pingbo Pan, Wei Chen, and Yi Yang. 2019. Dm-gan: Dynamic memory generative adversarial networks for text-to-image synthesis. In CVPR."},{"key":"e_1_3_2_1_51_1","volume-title":"CogCartoon: Towards Practical Story Visualization. arXiv preprint arXiv:2312.10718","author":"Zhu Zhongyang","year":"2023","unstructured":"Zhongyang Zhu and Jie Tang. 2023. CogCartoon: Towards Practical Story Visualization. arXiv preprint arXiv:2312.10718 (2023). n"}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Melbourne VIC Australia","acronym":"MM '24"},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680873","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3680873","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:08Z","timestamp":1750295888000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3680873"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":51,"alternative-id":["10.1145\/3664647.3680873","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3680873","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}