{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,17]],"date-time":"2026-07-17T06:06:09Z","timestamp":1784268369059,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":56,"publisher":"ACM","funder":[{"name":"National Nature Science Foundation of China","award":["62402406"],"award-info":[{"award-number":["62402406"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,15]]},"DOI":"10.1145\/3757377.3763833","type":"proceedings-article","created":{"date-parts":[[2025,12,8]],"date-time":"2025-12-08T16:27:29Z","timestamp":1765211249000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Context as Memory: Scene-Consistent Interactive Long Video Generation with Memory Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-8577-183X","authenticated-orcid":false,"given":"Jiwen","family":"Yu","sequence":"first","affiliation":[{"name":"University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3121-7259","authenticated-orcid":false,"given":"Jianhong","family":"Bai","sequence":"additional","affiliation":[{"name":"Zhejiang University, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-4561-0685","authenticated-orcid":false,"given":"Yiran","family":"Qin","sequence":"additional","affiliation":[{"name":"Chinese University of Hong Kong, Shenzhen, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3921-5960","authenticated-orcid":false,"given":"Quande","family":"Liu","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6585-8604","authenticated-orcid":false,"given":"Xintao","family":"Wang","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7225-565X","authenticated-orcid":false,"given":"Pengfei","family":"Wan","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-5475-2728","authenticated-orcid":false,"given":"Di","family":"Zhang","sequence":"additional","affiliation":[{"name":"Kuaishou Technology, Shenzhen, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1831-9952","authenticated-orcid":false,"given":"Xihui","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Hong Kong, Hong Kong, Hong Kong"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,12,14]]},"reference":[{"key":"e_1_3_3_2_2_1","unstructured":"Jianhong Bai Menghan Xia Xiao Fu Xintao Wang Lianrui Mu Jinwen Cao Zuozhu Liu Haoji Hu Xiang Bai Pengfei Wan et\u00a0al. 2025. ReCamMaster: Camera-Controlled Generative Rendering from A Single Video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.11647 (2025)."},{"key":"e_1_3_3_2_3_1","unstructured":"Jianhong Bai Menghan Xia Xintao Wang Ziyang Yuan Xiao Fu Zuozhu Liu Haoji Hu Pengfei Wan and Di Zhang. 2024. SynCamMaster: Synchronizing Multi-Camera Video Generation from Diverse Viewpoints. arxiv:https:\/\/arXiv.org\/abs\/2412.07760\u00a0[cs.CV] https:\/\/arxiv.org\/abs\/2412.07760"},{"key":"e_1_3_3_2_4_1","unstructured":"Fan Bao Chendong Xiang Gang Yue Guande He Hongzhou Zhu Kaiwen Zheng Min Zhao Shilong Liu Yaole Wang and Jun Zhu. 2024. Vidu: a highly consistent dynamic and skilled text-to-video generator with diffusion models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.04233 (2024)."},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"crossref","unstructured":"Boyuan Chen Diego\u00a0Marti Monso Yilun Du Max Simchowitz Russ Tedrake and Vincent Sitzmann. 2024. Diffusion forcing: Next-token prediction meets full-sequence diffusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2407.01392 (2024).","DOI":"10.52202\/079017-0759"},{"key":"e_1_3_3_2_6_1","unstructured":"Etched Decart. 2024. Oasis: A Universe in a Transformer. https:\/\/oasis-model.github.io\/."},{"key":"e_1_3_3_2_7_1","unstructured":"Google DeepMind. 2024a. Genie 2: A large-scale foundation world model. https:\/\/deepmind.google\/discover\/blog\/genie-2-a-large-scale-foundation-world-model\/."},{"key":"e_1_3_3_2_8_1","unstructured":"Google DeepMind. 2024b. Veo 2: Our state-of-the-art video generation model. https:\/\/deepmind.google\/technologies\/veo\/veo-2\/."},{"key":"e_1_3_3_2_9_1","unstructured":"Haoge Deng Ting Pan Haiwen Diao Zhengxiong Luo Yufeng Cui Huchuan Lu Shiguang Shan Yonggang Qi and Xinlong Wang. 2024. Autoregressive Video Generation without Vector Quantization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.14169 (2024)."},{"key":"e_1_3_3_2_10_1","unstructured":"Ruili Feng Han Zhang Zhantao Yang Jie Xiao Zhilei Shu Zhiheng Liu Andy Zheng Yukun Huang Yu Liu and Hongyang Zhang. 2024. The Matrix: Infinite-Horizon World Generation with Real-Time Moving Control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.03568 (2024)."},{"key":"e_1_3_3_2_11_1","volume-title":"ICLR","author":"Fu Xiao","year":"2025","unstructured":"Xiao Fu, Xian Liu, Xintao Wang, Sida Peng, Menghan Xia, Xiaoyu Shi, Ziyang Yuan, Pengfei Wan, Di Zhang, and Dahua Lin. 2025. 3DTrajMaster: Mastering 3D Trajectory for Multi-Entity Motion in Video Generation. In ICLR."},{"key":"e_1_3_3_2_12_1","unstructured":"Shenyuan Gao Jiazhi Yang Li Chen Kashyap Chitta Yihang Qiu Andreas Geiger Jun Zhang and Hongyang Li. 2024. Vista: A generalizable driving world model with high fidelity and versatile controllability. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2405.17398 (2024)."},{"key":"e_1_3_3_2_13_1","unstructured":"Yuchao Gu Weijia Mao and Mike\u00a0Zheng Shou. 2025. Long-Context Autoregressive Video Modeling with Next-Frame Prediction. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.19325 (2025)."},{"key":"e_1_3_3_2_14_1","unstructured":"Yuwei Guo Ceyuan Yang Ziyan Yang Zhibei Ma Zhijie Lin Zhenheng Yang Dahua Lin and Lu Jiang. 2025. Long context tuning for video generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.10589 (2025)."},{"key":"e_1_3_3_2_15_1","unstructured":"Hao He Yinghao Xu Yuwei Guo Gordon Wetzstein Bo Dai Hongsheng Li and Ceyuan Yang. 2024. Cameractrl: Enabling camera control for text-to-video generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.02101 (2024)."},{"key":"e_1_3_3_2_16_1","unstructured":"Jonathan Ho Ajay Jain and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. Advances in neural information processing systems (2020)."},{"key":"e_1_3_3_2_17_1","unstructured":"Jonathan Ho and Tim Salimans. 2022. Classifier-free diffusion guidance. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2207.12598 (2022)."},{"key":"e_1_3_3_2_18_1","unstructured":"Anthony Hu Lloyd Russell Hudson Yeo Zak Murez George Fedoseev Alex Kendall Jamie Shotton and Gianluca Corrado. 2023. Gaia-1: A generative world model for autonomous driving. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.17080 (2023)."},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"crossref","unstructured":"Anssi Kanervisto Dave Bignell Linda\u00a0Yilin Wen Martin Grayson Raluca Georgescu Sergio Valcarcel\u00a0Macua Shan\u00a0Zheng Tan Tabish Rashid Tim Pearce Yuhan Cao et\u00a0al. 2025. World and human action models towards gameplay ideation. Nature 638 8051 (2025) 656\u2013663.","DOI":"10.1038\/s41586-025-08600-3"},{"key":"e_1_3_3_2_20_1","unstructured":"Diederik\u00a0P Kingma Max Welling et\u00a0al. 2013. Auto-encoding variational bayes."},{"key":"e_1_3_3_2_21_1","unstructured":"Kling. 2024. Kling AI: Next-Generation AI Creative Studio. https:\/\/app.klingai.com\/."},{"key":"e_1_3_3_2_22_1","unstructured":"Dan Kondratyuk Lijun Yu Xiuye Gu Jos\u00e9 Lezama Jonathan Huang Grant Schindler Rachel Hornung Vighnesh Birodkar Jimmy Yan Ming-Chang Chiu et\u00a0al. 2023. Videopoet: A large language model for zero-shot video generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.14125 (2023)."},{"key":"e_1_3_3_2_23_1","unstructured":"Weijie Kong Qi Tian Zijian Zhang Rox Min Zuozhuo Dai Jin Zhou Jiangfeng Xiong Xin Li Bo Wu Jianwei Zhang et\u00a0al. 2024. Hunyuanvideo: A systematic framework for large video generative models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.03603 (2024)."},{"key":"e_1_3_3_2_24_1","unstructured":"Tianhong Li Yonglong Tian He Li Mingyang Deng and Kaiming He. 2024. Autoregressive Image Generation without Vector Quantization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2406.11838 (2024)."},{"key":"e_1_3_3_2_25_1","unstructured":"Yaron Lipman Ricky\u00a0TQ Chen Heli Ben-Hamu Maximilian Nickel and Matt Le. 2022. Flow matching for generative modeling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2210.02747 (2022)."},{"key":"e_1_3_3_2_26_1","unstructured":"Xingchao Liu Chengyue Gong and Qiang Liu. 2022. Flow straight and fast: Learning to generate and transfer data with rectified flow. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2209.03003 (2022)."},{"key":"e_1_3_3_2_27_1","unstructured":"Baorui Ma Huachen Gao Haoge Deng Zhengxiong Luo Tiejun Huang Lulu Tang and Xinlong Wang. 2024. You See it You Got it: Learning 3D Creation on Pose-Free Videos at Scale. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2412.06699 (2024)."},{"key":"e_1_3_3_2_28_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i5.28226"},{"key":"e_1_3_3_2_29_1","unstructured":"OpenAI. 2024. Creating video from text. https:\/\/openai.com\/index\/sora\/."},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00387"},{"key":"e_1_3_3_2_31_1","unstructured":"Yiran Qin Zhelun Shi Jiwen Yu Xijun Wang Enshen Zhou Lijun Li Zhenfei Yin Xihui Liu Lu Sheng Jing Shao et\u00a0al. 2024. Worldsimbench: Towards video generation models as world simulators. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2410.18072 (2024)."},{"key":"e_1_3_3_2_32_1","unstructured":"Xuanchi Ren Tianchang Shen Jiahui Huang Huan Ling Yifan Lu Merlin Nimier-David Thomas M\u00fcller Alexander Keller Sanja Fidler and Jun Gao. 2025. Gen3c: 3d-informed world-consistent video generation with precise camera control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.03751 (2025)."},{"key":"e_1_3_3_2_33_1","unstructured":"Runway. 2024. Runway : Tools for human imagination. https:\/\/runwayml.com\/."},{"key":"e_1_3_3_2_34_1","unstructured":"Lloyd Russell Anthony Hu Lorenzo Bertoni George Fedoseev Jamie Shotton Elahe Arani and Gianluca Corrado. 2025. GAIA-2: A Controllable Multi-View Generative World Model for Autonomous Driving. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.20523 (2025)."},{"key":"e_1_3_3_2_35_1","unstructured":"Kiwhan Song Boyuan Chen Max Simchowitz Yilun Du Russ Tedrake and Vincent Sitzmann. 2025. History-Guided Video Diffusion. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2502.06764 (2025)."},{"key":"e_1_3_3_2_36_1","unstructured":"Yang Song and Stefano Ermon. 2019. Generative modeling by estimating gradients of the data distribution. Advances in neural information processing systems (2019)."},{"key":"e_1_3_3_2_37_1","unstructured":"Yang Song Jascha Sohl-Dickstein Diederik\u00a0P Kingma Abhishek Kumar Stefano Ermon and Ben Poole. 2021. Score-based generative modeling through stochastic differential equations. International Conference on Learning Representations (2021)."},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"crossref","unstructured":"Jianlin Su Murtadha Ahmed Yu Lu Shengfeng Pan Wen Bo and Yunfeng Liu. 2024. Roformer: Enhanced transformer with rotary position embedding. Neurocomputing 568 (2024) 127063.","DOI":"10.1016\/j.neucom.2023.127063"},{"key":"e_1_3_3_2_39_1","unstructured":"Dani Valevski Yaniv Leviathan Moab Arar and Shlomi Fruchter. 2024. Diffusion models are real-time game engines. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.14837 (2024)."},{"key":"e_1_3_3_2_40_1","unstructured":"Ang Wang Baole Ai Bin Wen Chaojie Mao Chen-Wei Xie Di Chen Feiwu Yu Haiming Zhao Jianxiao Yang Jianyuan Zeng et\u00a0al. 2025. Wan: Open and Advanced Large-Scale Video Generative Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.20314 (2025)."},{"key":"e_1_3_3_2_41_1","unstructured":"Xinlong Wang Xiaosong Zhang Zhengxiong Luo Quan Sun Yufeng Cui Jinsheng Wang Fan Zhang Yueze Wang Zhen Li Qiying Yu et\u00a0al. 2024b. Emu3: Next-token prediction is all you need. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.18869 (2024)."},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"publisher","DOI":"10.1145\/3641519.3657518"},{"key":"e_1_3_3_2_43_1","unstructured":"Zeqi Xiao Yushi Lan Yifan Zhou Wenqi Ouyang Shuai Yang Yanhong Zeng and Xingang Pan. 2025. WORLDMEM: Long-term Consistent World Simulation with Memory. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.12369 (2025)."},{"key":"e_1_3_3_2_44_1","unstructured":"Jinbo Xing Menghan Xia Yong Zhang Haoxin Chen Xintao Wang Tien-Tsin Wong and Ying Shan. 2023. DynamiCrafter: Animating Open-domain Images with Video Diffusion Priors. arxiv:https:\/\/arXiv.org\/abs\/2310.12190"},{"key":"e_1_3_3_2_45_1","unstructured":"Wilson Yan Yunzhi Zhang Pieter Abbeel and Aravind Srinivas. 2021. Videogpt: Video generation using vq-vae and transformers. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2104.10157 (2021)."},{"key":"e_1_3_3_2_46_1","unstructured":"Mengjiao Yang Yilun Du Kamyar Ghasemipour Jonathan Tompson Dale Schuurmans and Pieter Abbeel. 2023. Learning Interactive Real-World Simulators. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.06114 (2023)."},{"key":"e_1_3_3_2_47_1","volume-title":"Proceedings of the 41st International Conference on Machine Learning","author":"Yang Sherry","year":"2024","unstructured":"Sherry Yang, Jacob\u00a0C Walker, Jack Parker-Holder, Yilun Du, Jake Bruce, Andre Barreto, Pieter Abbeel, and Dale Schuurmans. 2024b. Position: Video as the New Language for Real-World Decision Making. In Proceedings of the 41st International Conference on Machine Learning."},{"key":"e_1_3_3_2_48_1","unstructured":"Zhuoyi Yang Jiayan Teng Wendi Zheng Ming Ding Shiyu Huang Jiazheng Xu Yuanming Yang Wenyi Hong Xiaohan Zhang Guanyu Feng et\u00a0al. 2024a. CogVideoX: Text-to-Video Diffusion Models with An Expert Transformer. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.06072 (2024)."},{"key":"e_1_3_3_2_49_1","unstructured":"Yuan Yao Tianyu Yu Ao Zhang Chongyi Wang Junbo Cui Hongji Zhu Tianchi Cai Haoyu Li Weilin Zhao Zhihui He et\u00a0al. 2024. MiniCPM-V: A GPT-4V Level MLLM on Your Phone. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2408.01800 (2024)."},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.00636"},{"key":"e_1_3_3_2_51_1","unstructured":"Jiwen Yu Yiran Qin Haoxuan Che Quande Liu Xintao Wang Pengfei Wan Di Zhang Kun Gai Hao Chen and Xihui Liu. 2025b. A Survey of Interactive Generative Video. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.21853 (2025)."},{"key":"e_1_3_3_2_52_1","unstructured":"Jiwen Yu Yiran Qin Haoxuan Che Quande Liu Xintao Wang Pengfei Wan Di Zhang and Xihui Liu. 2025a. Position: Interactive generative video as next-generation game engine. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2503.17359 (2025)."},{"key":"e_1_3_3_2_53_1","unstructured":"Jiwen Yu Yiran Qin Xintao Wang Pengfei Wan Di Zhang and Xihui Liu. 2025c. GameFactory: Creating New Games with Generative Interactive Videos. arxiv:https:\/\/arXiv.org\/abs\/2501.08325"},{"key":"e_1_3_3_2_54_1","unstructured":"Wangbo Yu Jinbo Xing Li Yuan Wenbo Hu Xiaoyu Li Zhipeng Huang Xiangjun Gao Tien-Tsin Wong Ying Shan and Yonghong Tian. 2024b. Viewcrafter: Taming video diffusion models for high-fidelity novel view synthesis. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2409.02048 (2024)."},{"key":"e_1_3_3_2_55_1","unstructured":"Lvmin Zhang and Maneesh Agrawala. 2025. Packing Input Frame Context in Next-Frame Prediction Models for Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2504.12626 (2025)."},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00355"},{"key":"e_1_3_3_2_57_1","unstructured":"Tinghui Zhou Richard Tucker John Flynn Graham Fyffe and Noah Snavely. 2018. Stereo magnification: Learning view synthesis using multiplane images. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1805.09817 (2018)."}],"event":{"name":"SA Conference Papers '25: SIGGRAPH Asia 2025 Conference Papers","location":"Hong Kong Hong Kong","acronym":"SA Conference Papers '25","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["Proceedings of the SIGGRAPH Asia 2025 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3757377.3763833","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T03:25:13Z","timestamp":1765250713000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3757377.3763833"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,14]]},"references-count":56,"alternative-id":["10.1145\/3757377.3763833","10.1145\/3757377"],"URL":"https:\/\/doi.org\/10.1145\/3757377.3763833","relation":{},"subject":[],"published":{"date-parts":[[2025,12,14]]},"assertion":[{"value":"2025-12-14","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}