{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T14:31:38Z","timestamp":1773930698863,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":63,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,11,21]],"date-time":"2024-11-21T00:00:00Z","timestamp":1732147200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,11,21]]},"DOI":"10.1145\/3677388.3696321","type":"proceedings-article","created":{"date-parts":[[2024,10,30]],"date-time":"2024-10-30T08:12:19Z","timestamp":1730275939000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":3,"title":["From Words to Worlds: Transforming One-line Prompts into Multi-modal Digital Stories with LLM Agents"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-5041-2079","authenticated-orcid":false,"given":"Danrui","family":"Li","sequence":"first","affiliation":[{"name":"Computer Science, Rutgers, the State University of New Jersey, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4700-954X","authenticated-orcid":false,"given":"Samuel S.","family":"Sohn","sequence":"additional","affiliation":[{"name":"Computer Science, Rutgers, the State University of New Jersey, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8445-4848","authenticated-orcid":false,"given":"Sen","family":"Zhang","sequence":"additional","affiliation":[{"name":"Computer Science, Rutgers, the State University of New Jersey, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7935-8723","authenticated-orcid":false,"given":"Che-Jui","family":"Chang","sequence":"additional","affiliation":[{"name":"Computer Science, Rutgers, the State University of New Jersey, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3501-0028","authenticated-orcid":false,"given":"Mubbasir","family":"Kapadia","sequence":"additional","affiliation":[{"name":"Roblox, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,11,21]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6232"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/TG.2022.3216582"},{"key":"e_1_3_2_2_3_1","unstructured":"James Betker Gabriel Goh Li Jing Tim Brooks Jianfeng Wang Linjie Li Long Ouyang Juntang Zhuang Joyce Lee Yufei Guo Wesam Manassra Prafulla Dhariwal Casey Chu Yunxin Jiao and Aditya Ramesh. 2023. Improving Image Generation with Better Captions. https:\/\/cdn.openai.com\/papers\/dall-e-3.pdf"},{"key":"e_1_3_2_2_4_1","volume-title":"Zoedepth: Zero-shot transfer by combining relative and metric depth. arXiv preprint arXiv:2302.12288","author":"Bhat Shariq\u00a0Farooq","year":"2023","unstructured":"Shariq\u00a0Farooq Bhat, Reiner Birkl, Diana Wofk, Peter Wonka, and Matthias M\u00fcller. 2023a. Zoedepth: Zero-shot transfer by combining relative and metric depth. arXiv preprint arXiv:2302.12288 (2023)."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","unstructured":"Shariq\u00a0Farooq Bhat Reiner Birkl Diana Wofk Peter Wonka and Matthias M\u00fcller. 2023b. ZoeDepth: Zero-shot Transfer by Combining Relative and Metric Depth. https:\/\/doi.org\/10.48550\/ARXIV.2302.12288","DOI":"10.48550\/ARXIV.2302.12288"},{"key":"e_1_3_2_2_6_1","unstructured":"Tim Brooks Bill Peebles Connor Holmes Will DePue Yufei Guo Li Jing David Schnurr Joe Taylor Troy Luhman Eric Luhman Clarence Ng Ricky Wang and Aditya Ramesh. 2024. Video generation models as world simulators. (2024). https:\/\/openai.com\/research\/video-generation-models-as-world-simulators"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/1291233.1291387"},{"key":"e_1_3_2_2_8_1","volume-title":"On the Equivalency, Substitutability, and Flexibility of Synthetic Data. arXiv preprint arXiv:2403.16244","author":"Chang Che-Jui","year":"2024","unstructured":"Che-Jui Chang, Danrui Li, Seonghyeon Moon, and Mubbasir Kapadia. 2024a. On the Equivalency, Substitutability, and Flexibility of Synthetic Data. arXiv preprint arXiv:2403.16244 (2024)."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02070"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581641.3584045"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1145\/3536221.3558060"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1002\/cav.2076"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1002\/cav.1960"},{"key":"e_1_3_2_2_14_1","unstructured":"CoquiAI. 2021. coqui-ai\/TTS: a deep learning toolkit for Text-to-Speech battle-tested in research and production. https:\/\/github.com\/coqui-ai\/TTS?tab=readme-ov-file"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1080\/15213269.2015.1015740"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"crossref","unstructured":"Adele De\u00a0Jager Andrea Fogarty Anna Tewson Caroline Lenette and Katherine\u00a0M Boydell. 2017. Digital storytelling in research: A systematic review. The Qualitative Report 22 10 (2017) 2548\u20132582.","DOI":"10.46743\/2160-3715\/2017.2970"},{"key":"e_1_3_2_2_17_1","unstructured":"ElevenLabs. 2024. elevenlabs\/elevenlabs-python. https:\/\/github.com\/elevenlabs\/elevenlabs-python original-date: 2023-03-26T11:59:52Z."},{"key":"e_1_3_2_2_18_1","unstructured":"Patrick Esser Sumith Kulal Andreas Blattmann Rahim Entezari Jonas M\u00fcller Harry Saini Yam Levi Dominik Lorenz Axel Sauer Frederic Boesel Dustin Podell Tim Dockhorn Zion English Kyle Lacey Alex Goodwin Yannik Marek and Robin Rombach. 2024. Scaling Rectified Flow Transformers for High-Resolution Image Synthesis. arxiv:2403.03206\u00a0[cs.CV]"},{"key":"e_1_3_2_2_19_1","unstructured":"Mengyang Feng Jinlin Liu Kai Yu Yuan Yao Zheng Hui Xiefan Guo Xianhui Lin Haolan Xue Chen Shi Xiaowen Li Aojie Li Xiaoyang Kang Biwen Lei Miaomiao Cui Peiran Ren and Xuansong Xie. 2023. DreaMoving: A Human Video Generation Framework based on Diffusion Models. arxiv:2312.05107\u00a0[cs.CV]"},{"key":"e_1_3_2_2_20_1","unstructured":"Yuwei Guo Ceyuan Yang Anyi Rao Zhengyang Liang Yaohui Wang Yu Qiao Maneesh Agrawala Dahua Lin and Bo Dai. 2023. AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/CIG.2011.6032020"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1145\/3588432.3591525"},{"key":"e_1_3_2_2_23_1","volume-title":"Imagen Video: High Definition Video Generation with Diffusion Models. arxiv:2210.02303\u00a0[cs.CV]","author":"Ho Jonathan","year":"2022","unstructured":"Jonathan Ho, William Chan, Chitwan Saharia, Jay Whang, Ruiqi Gao, Alexey Gritsenko, Diederik\u00a0P. Kingma, Ben Poole, Mohammad Norouzi, David\u00a0J. Fleet, and Tim Salimans. 2022. Imagen Video: High Definition Video Generation with Diffusion Models. arxiv:2210.02303\u00a0[cs.CV]"},{"key":"e_1_3_2_2_24_1","volume-title":"Proceedings of the ACM SIGGRAPH\/Eurographics Symposium on Computer Animation","author":"Kapadia Mubbasir","year":"2016","unstructured":"Mubbasir Kapadia, Seth Frey, Alexander Shoulson, Robert\u00a0W. Sumner, and Markus Gross. 2016a. CANVAS: computer-assisted narrative animation synthesis. In Proceedings of the ACM SIGGRAPH\/Eurographics Symposium on Computer Animation (Zurich, Switzerland) (SCA \u201916). Eurographics Association, Goslar, DEU, 199\u2013209."},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1145\/2994258.2994265"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.3390\/electronics12061289"},{"key":"e_1_3_2_2_27_1","unstructured":"Felix Kreuk Gabriel Synnaeve Adam Polyak Uriel Singer Alexandre D\u00e9fossez Jade Copet Devi Parikh Yaniv Taigman and Yossi Adi. 2023. AudioGen: Textually Guided Audio Generation. arxiv:2209.15352\u00a0[cs.SD]"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1609\/aiide.v19i1.27504"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCIAIG.2016.2546063"},{"key":"e_1_3_2_2_30_1","volume-title":"Digital storytelling: Capturing lives, creating community","author":"Lambert Joe","unstructured":"Joe Lambert. 2013. Digital storytelling: Capturing lives, creating community. Routledge."},{"key":"e_1_3_2_2_31_1","unstructured":"Daiqing Li Aleks Kamko Ehsan Akhgari Ali Sabet Linmiao Xu and Suhail Doshi. 2024. Playground v2.5: Three Insights towards Enhancing Aesthetic Quality in Text-to-Image Generation. arxiv:2402.17245\u00a0[cs.CV]"},{"key":"e_1_3_2_2_32_1","volume-title":"Thirty-seventh Conference on Neural Information Processing Systems.","author":"Li Guohao","year":"2023","unstructured":"Guohao Li, Hasan Abed Al\u00a0Kader Hammoud, Hani Itani, Dmitrii Khizbullin, and Bernard Ghanem. 2023. CAMEL: Communicative Agents for \"Mind\" Exploration of Large Language Model Society. In Thirty-seventh Conference on Neural Information Processing Systems."},{"key":"e_1_3_2_2_33_1","unstructured":"Jun\u00a0Hao Liew Hanshu Yan Jianfeng Zhang Zhongcong Xu and Jiashi Feng. 2023. MagicEdit: High-Fidelity and Temporally Coherent Video Editing. arxiv:2308.14749\u00a0[cs.CV]"},{"key":"e_1_3_2_2_34_1","unstructured":"Bo Liu Yuqian Jiang Xiaohan Zhang Qiang Liu Shiqi Zhang Joydeep Biswas and Peter Stone. 2023a. LLM+P: Empowering Large Language Models with Optimal Planning Proficiency. arxiv:2304.11477\u00a0[cs.AI] https:\/\/arxiv.org\/abs\/2304.11477"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"crossref","unstructured":"Chang Liu Haoning Wu Yujie Zhong Xiaoyun Zhang Yanfeng Wang and Weidi Xie. 2024. Intelligent Grimm \u2013 Open-ended Visual Storytelling via Latent Diffusion Models. arxiv:2306.00973\u00a0[cs.CV]","DOI":"10.1109\/CVPR52733.2024.00592"},{"key":"e_1_3_2_2_36_1","unstructured":"Xubo Liu Zhongkai Zhu Haohe Liu Yi Yuan Meng Cui Qiushi Huang Jinhua Liang Yin Cao Qiuqiang Kong Mark\u00a0D. Plumbley and Wenwu Wang. 2023b. WavJourney: Compositional Audio Creation with Large Language Models. arxiv:2307.14335\u00a0[cs.SD]"},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1145\/3274247.3274500"},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-031-19836-6_5"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1145\/3172944.3172972"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1609\/aiide.v19i1.27506"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1145\/3337722.3341850"},{"key":"e_1_3_2_2_42_1","unstructured":"Music Technology\u00a0Group of Universitat Pompeu\u00a0Fabra. [n. d.]. Freesound. https:\/\/freesound.org\/"},{"key":"e_1_3_2_2_43_1","volume-title":"Generative Agents: Interactive Simulacra of Human Behavior. arxiv:2304.03442\u00a0[cs.HC]","author":"Park Joon\u00a0Sung","year":"2023","unstructured":"Joon\u00a0Sung Park, Joseph\u00a0C. O\u2019Brien, Carrie\u00a0J. Cai, Meredith\u00a0Ringel Morris, Percy Liang, and Michael\u00a0S. Bernstein. 2023. Generative Agents: Interactive Simulacra of Human Behavior. arxiv:2304.03442\u00a0[cs.HC]"},{"key":"e_1_3_2_2_44_1","unstructured":"Chen Qian Xin Cong Wei Liu Cheng Yang Weize Chen Yusheng Su Yufan Dang Jiahao Li Juyuan Xu Dahai Li Zhiyuan Liu and Maosong Sun. 2023. Communicative Agents for Software Development. arxiv:2307.07924\u00a0[cs.SE]"},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610543.3626176"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1145\/3356590.3356619"},{"key":"e_1_3_2_2_47_1","unstructured":"Xiaoqian Shen and Mohamed Elhoseiny. 2023. StoryGPT-V: Large Language Models as Consistent Story Visualizers. arxiv:2312.02252\u00a0[cs.CV]"},{"key":"e_1_3_2_2_48_1","unstructured":"Simpleton. 2021. CC2D Essential Bundle | 2D Characters | Unity Asset Store. https:\/\/assetstore.unity.com\/packages\/2d\/characters\/cc2d-essential-bundle-187410"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1145\/3355089.3356505"},{"key":"e_1_3_2_2_50_1","unstructured":"Damian Stewart. [n. d.]. Compel: A prompting enhancement library for transformers-type text embedding systems. https:\/\/github.com\/damian0815\/compel"},{"key":"e_1_3_2_2_51_1","volume-title":"LLMR: Real-time Prompting of Interactive Worlds using Large Language Models. arxiv:2309.12276\u00a0[cs.HC]","author":"De\u00a0La Torre Fernanda","year":"2024","unstructured":"Fernanda De\u00a0La Torre, Cathy\u00a0Mengying Fang, Han Huang, Andrzej Banburski-Fahey, Judith\u00a0Amores Fernandez, and Jaron Lanier. 2024. LLMR: Real-time Prompting of Interactive Worlds using Large Language Models. arxiv:2309.12276\u00a0[cs.HC]"},{"key":"e_1_3_2_2_52_1","volume-title":"Chi, Quoc Le, and Denny Zhou","author":"Wei Jason","year":"2023","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed Chi, Quoc Le, and Denny Zhou. 2023. Chain-of-Thought Prompting Elicits Reasoning in Large Language Models. arxiv:2201.11903\u00a0[cs.CL]"},{"key":"e_1_3_2_2_53_1","volume-title":"Proceedings of the 36th International Conference on Neural Information Processing Systems","author":"Wei Jason","year":"2024","unstructured":"Jason Wei, Xuezhi Wang, Dale Schuurmans, Maarten Bosma, Brian Ichter, Fei Xia, Ed\u00a0H. Chi, Quoc\u00a0V. Le, and Denny Zhou. 2024. Chain-of-thought prompting elicits reasoning in large language models. In Proceedings of the 36th International Conference on Neural Information Processing Systems (New Orleans, LA, USA) (NIPS \u201922). Curran Associates Inc., Red Hook, NY, USA, Article 1800, 14\u00a0pages."},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.compedu.2019.103786"},{"key":"e_1_3_2_2_55_1","unstructured":"Qingyun Wu Gagan Bansal Jieyu Zhang Yiran Wu Beibin Li Erkang Zhu Li Jiang Xiaoyun Zhang Shaokun Zhang Jiale Liu Ahmed\u00a0Hassan Awadallah Ryen\u00a0W White Doug Burger and Chi Wang. 2023. AutoGen: Enabling Next-Gen LLM Applications via Multi-Agent Conversation. arxiv:2308.08155\u00a0[cs.AI]"},{"key":"e_1_3_2_2_56_1","unstructured":"Enze Xie Wenhai Wang Zhiding Yu Anima Anandkumar Jose\u00a0M Alvarez and Ping Luo. 2021. SegFormer: Simple and Efficient Design for Semantic Segmentation with Transformers. In Neural Information Processing Systems (NeurIPS)."},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1145\/3340531.3411937"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1145\/3618375"},{"key":"e_1_3_2_2_59_1","volume-title":"Tree of Thoughts: Deliberate Problem Solving with Large Language Models. Advances in Neural Information Processing Systems 36","author":"Yao Shunyu","year":"2023","unstructured":"Shunyu Yao, Dian Yu, Jeffrey Zhao, Izhak Shafran, Thomas L. Griffiths, Yuan Cao, and Karthik Narasimhan. 2023a. Tree of Thoughts: Deliberate Problem Solving with Large Language Models. Advances in Neural Information Processing Systems 36 (2023). Publisher Copyright: \u00a9 2023 Neural information processing systems foundation. All rights reserved.; 37th Conference on Neural Information Processing Systems, NeurIPS 2023 ; Conference date: 10-12-2023 Through 16-12-2023."},{"key":"e_1_3_2_2_60_1","unstructured":"Shunyu Yao Jeffrey Zhao Dian Yu Nan Du Izhak Shafran Karthik Narasimhan and Yuan Cao. 2023b. ReAct: Synergizing Reasoning and Acting in Language Models. arxiv:2210.03629\u00a0[cs.CL]"},{"key":"e_1_3_2_2_61_1","unstructured":"Hongwei Yi Justus Thies Michael\u00a0J. Black Xue\u00a0Bin Peng and Davis Rempe. 2024. Generating Human Interaction Motions in Scenes with Text Control. arxiv:2404.10685\u00a0[cs.CV]"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1111\/cgf.14415"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"crossref","unstructured":"Zhexin Zhang Jiaxin Wen Jian Guan and Minlie Huang. 2022. Persona-Guided Planning for Controlling the Protagonist\u2019s Persona in Story Generation. arxiv:2204.10703\u00a0[cs.CL]","DOI":"10.18653\/v1\/2022.naacl-main.245"}],"event":{"name":"MIG '24: The 17th ACM SIGGRAPH Conference on Motion, Interaction, and Games","location":"Arlington VA USA","acronym":"MIG '24","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["The 17th ACM SIGGRAPH Conference on Motion Interaction and Games"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3677388.3696321","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3677388.3696321","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,8,29]],"date-time":"2025-08-29T13:07:22Z","timestamp":1756472842000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3677388.3696321"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,11,21]]},"references-count":63,"alternative-id":["10.1145\/3677388.3696321","10.1145\/3677388"],"URL":"https:\/\/doi.org\/10.1145\/3677388.3696321","relation":{},"subject":[],"published":{"date-parts":[[2024,11,21]]},"assertion":[{"value":"2024-11-21","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}