{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:49:32Z","timestamp":1778082572865,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":68,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T00:00:00Z","timestamp":1733184000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,12,3]]},"DOI":"10.1145\/3680528.3687656","type":"proceedings-article","created":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T08:14:37Z","timestamp":1733213677000},"page":"1-11","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":23,"title":["I2VEdit: First-Frame-Guided Video Editing via Image-to-Video Diffusion Models"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-2273-5984","authenticated-orcid":false,"given":"Wenqi","family":"Ouyang","sequence":"first","affiliation":[{"name":"S-Lab, Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-6017-2886","authenticated-orcid":false,"given":"Yi","family":"Dong","sequence":"additional","affiliation":[{"name":"Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0571-5924","authenticated-orcid":false,"given":"Lei","family":"Yang","sequence":"additional","affiliation":[{"name":"SenseTime Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2029-6588","authenticated-orcid":false,"given":"Jianlou","family":"Si","sequence":"additional","affiliation":[{"name":"SenseTime Research, Beijing, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5825-9467","authenticated-orcid":false,"given":"Xingang","family":"Pan","sequence":"additional","affiliation":[{"name":"S-Lab, Nanyang Technological University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,12,3]]},"reference":[{"key":"e_1_3_3_2_2_1","unstructured":"2023. Gen-2 by runway. https:\/\/research.runwayml.com\/gen2."},{"key":"e_1_3_3_2_3_1","unstructured":"2023. Pika labs. https:\/\/pika.art\/."},{"key":"e_1_3_3_2_4_1","volume-title":"Adobe Photoshop","author":"Inc. Adobe","unstructured":"Adobe Inc.[n. d.]. Adobe Photoshop. https:\/\/www.adobe.com\/products\/photoshop.html"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"publisher","unstructured":"Xiaobo An and Fabio Pellacini. 2008. AppProp: all-pairs appearance-space edit propagation. ACM Trans. Graph. 27 3 (aug 2008) 1\u20139. 10.1145\/1360612.1360639https:\/\/dl.acm.org\/doi\/10.1145\/1360612.1360639","DOI":"10.1145\/1360612.1360639"},{"key":"e_1_3_3_2_6_1","doi-asserted-by":"crossref","unstructured":"Theodore\u00a0W Anderson and Donald\u00a0A Darling. 1954. A test of goodness of fit. Journal of the American statistical association 49 268 (1954) 765\u2013769.","DOI":"10.1080\/01621459.1954.10501232"},{"key":"e_1_3_3_2_7_1","unstructured":"Andreas Blattmann Tim Dockhorn Sumith Kulal Daniel Mendelevitch Maciej Kilian Dominik Lorenz Yam Levi Zion English Vikram Voleti Adam Letts et\u00a0al. 2023. Stable video diffusion: Scaling latent video diffusion models to large datasets. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.15127 (2023)."},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"crossref","unstructured":"Tim Brooks Aleksander Holynski and Alexei\u00a0A Efros. 2022. InstructPix2Pix: Learning to Follow Image Editing Instructions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2211.09800 (2022).","DOI":"10.1109\/CVPR52729.2023.01764"},{"key":"e_1_3_3_2_9_1","unstructured":"Tim Brooks Bill Peebles Connor Holmes Will DePue Yufei Guo Li Jing David Schnurr Joe Taylor Troy Luhman Eric Luhman Clarence Ng Ricky Wang and Aditya Ramesh. 2024. Video generation models as world simulators. https:\/\/openai.com\/research\/video-generation-models-as-world-simulators"},{"key":"e_1_3_3_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.02062"},{"key":"e_1_3_3_2_11_1","unstructured":"Xi Chen Lianghua Huang Yu Liu Yujun Shen Deli Zhao and Hengshuang Zhao. 2023b. Anydoor: Zero-shot object-level image customization. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2307.09481 (2023)."},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"crossref","unstructured":"Yutao Chen Xingning Dong Tian Gan Chunluan Zhou Ming Yang and Qingpei Guo. 2023a. Eve: Efficient zero-shot text-based video editing with depth map guidance and temporal consistency constraints. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.10648 (2023).","DOI":"10.24963\/ijcai.2024\/75"},{"key":"e_1_3_3_2_13_1","unstructured":"Yisol Choi Sangkyung Kwak Kyungmin Lee Hyungwon Choi and Jinwoo Shin. 2024. Improving Diffusion Models for Authentic Virtual Try-on in the Wild. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.05139 (2024)."},{"key":"e_1_3_3_2_14_1","unstructured":"Yuren Cong Mengmeng Xu Christian Simon Shoufa Chen Jiawei Ren Yanping Xie Juan-Manuel Perez-Rua Bodo Rosenhahn Tao Xiang and Sen He. 2023. FLATTEN: optical FLow-guided ATTENtion for consistent text-to-video editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.05922 (2023)."},{"key":"e_1_3_3_2_15_1","unstructured":"Guillaume Couairon Jakob Verbeek Holger Schwenk and Matthieu Cord. 2022. Diffedit: Diffusion-based semantic image editing with mask guidance. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2210.11427 (2022)."},{"key":"e_1_3_3_2_16_1","unstructured":"Ziyi Dong Pengxu Wei and Liang Lin. 2022. Dreamartist: Towards controllable one-shot text-to-image generation via contrastive prompt-tuning. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2211.11337 (2022)."},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612680"},{"key":"e_1_3_3_2_18_1","unstructured":"Michal Geyer Omer Bar-Tal Shai Bagon and Tali Dekel. 2023. TokenFlow: Consistent Diffusion Features for Consistent Video Editing. arXiv preprint arxiv:https:\/\/arXiv.org\/abs\/2307.10373 (2023)."},{"key":"e_1_3_3_2_19_1","unstructured":"Yuchao Gu Xintao Wang Jay\u00a0Zhangjie Wu Yujun Shi Chen Yunpeng Zihan Fan Wuyou Xiao Rui Zhao Shuning Chang Weijia Wu Yixiao Ge Shan Ying and Mike\u00a0Zheng Shou. 2023a. Mix-of-Show: Decentralized Low-Rank Adaptation for Multi-Concept Customization of Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.18292 (2023)."},{"key":"e_1_3_3_2_20_1","unstructured":"Yuchao Gu Yipin Zhou Bichen Wu Licheng Yu Jia-Wei Liu Rui Zhao Jay\u00a0Zhangjie Wu David\u00a0Junhao Zhang Mike\u00a0Zheng Shou and Kevin Tang. 2023b. VideoSwap: Customized Video Subject Swapping with Interactive Semantic Point Correspondence. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.02087 (2023)."},{"key":"e_1_3_3_2_21_1","unstructured":"Yuwei Guo Ceyuan Yang Anyi Rao Maneesh Agrawala Dahua Lin and Bo Dai. 2023. SparseCtrl: Adding Sparse Controls to Text-to-Video Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.16933 (2023)."},{"key":"e_1_3_3_2_22_1","unstructured":"Yuwei Guo Ceyuan Yang Anyi Rao Zhengyang Liang Yaohui Wang Yu Qiao Maneesh Agrawala Dahua Lin and Bo Dai. 2024. AnimateDiff: Animate Your Personalized Text-to-Image Diffusion Models without Specific Tuning. International Conference on Learning Representations (2024)."},{"key":"e_1_3_3_2_23_1","unstructured":"Amir Hertz Ron Mokady Jay Tenenbaum Kfir Aberman Yael Pritch and Daniel Cohen-Or. 2022. Prompt-to-Prompt Image Editing with Cross Attention Control. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2208.01626 (2022)."},{"key":"e_1_3_3_2_24_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.emnlp-main.595"},{"key":"e_1_3_3_2_25_1","unstructured":"Edward\u00a0J Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2021. Lora: Low-rank adaptation of large language models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2106.09685 (2021)."},{"key":"e_1_3_3_2_26_1","doi-asserted-by":"crossref","unstructured":"Ond\u0159ej Jamri\u0161ka \u0160\u00e1rka Sochorov\u00e1 Ond\u0159ej Texler Michal Luk\u00e1\u010d Jakub Fi\u0161er Jingwan Lu Eli Shechtman and Daniel S\u1ef3kora. 2019. Stylizing video by example. ACM Transactions on Graphics (TOG) 38 4 (2019) 1\u201311.","DOI":"10.1145\/3306346.3323006"},{"key":"e_1_3_3_2_27_1","unstructured":"Hyeonho Jeong Geon\u00a0Yeong Park and Jong\u00a0Chul Ye. 2023. VMC: Video Motion Customization using Temporal Attention Adaption for Text-to-Video Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2312.00845 (2023)."},{"key":"e_1_3_3_2_28_1","unstructured":"Hyeonho Jeong and Jong\u00a0Chul Ye. 2023. Ground-A-Video: Zero-shot Grounded Video Editing using Text-to-image Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.01107 (2023)."},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"crossref","unstructured":"Nick Kanopoulos Nagesh Vasanthavada and Robert\u00a0L Baker. 1988. Design of an image edge detection filter using the Sobel operator. IEEE Journal of solid-state circuits 23 2 (1988) 358\u2013367.","DOI":"10.1109\/4.996"},{"key":"e_1_3_3_2_30_1","volume-title":"Proc. NeurIPS","author":"Karras Tero","year":"2022","unstructured":"Tero Karras, Miika Aittala, Timo Aila, and Samuli Laine. 2022. Elucidating the Design Space of Diffusion-Based Generative Models. In Proc. NeurIPS."},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"crossref","unstructured":"Levon Khachatryan Andranik Movsisyan Vahram Tadevosyan Roberto Henschel Zhangyang Wang Shant Navasardyan and Humphrey Shi. 2023. Text2Video-Zero: Text-to-Image Diffusion Models are Zero-Shot Video Generators. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.13439 (2023).","DOI":"10.1109\/ICCV51070.2023.01462"},{"key":"e_1_3_3_2_32_1","unstructured":"Max Ku Cong Wei Weiming Ren Harry Yang and Wenhu Chen. 2024. AnyV2V: A Plug-and-Play Framework For Any Video-to-Video Editing Tasks. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2403.14468 (2024)."},{"key":"e_1_3_3_2_33_1","unstructured":"Sangyun Lee Gyojung Gu Sunghyun Park Seunghwan Choi and Jaegul Choo. 2022. High-Resolution Virtual Try-On with Misalignment and Occlusion-Handled Conditions. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2206.14180 (2022)."},{"key":"e_1_3_3_2_34_1","volume-title":"ICML","author":"Li Junnan","year":"2022","unstructured":"Junnan Li, Dongxu Li, Caiming Xiong, and Steven Hoi. 2022. BLIP: Bootstrapping Language-Image Pre-training for Unified Vision-Language Understanding and Generation. In ICML."},{"key":"e_1_3_3_2_35_1","unstructured":"Jingyun Liang Yuchen Fan Kai Zhang Radu Timofte Luc Van\u00a0Gool and Rakesh Ranjan. 2023. MoVideo: Motion-Aware Video Generation with Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.00000 (2023)."},{"key":"e_1_3_3_2_36_1","unstructured":"Shaoteng Liu Yuechen Zhang Wenbo Li Zhe Lin and Jiaya Jia. 2023. Video-P2P: Video Editing with Cross-attention Control."},{"key":"e_1_3_3_2_37_1","unstructured":"Ron Mokady Amir Hertz Kfir Aberman Yael Pritch and Daniel Cohen-Or. 2022. Null-text Inversion for Editing Real Images using Guided Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2211.09794 (2022)."},{"key":"e_1_3_3_2_38_1","unstructured":"Hao Ouyang Qiuyu Wang Yuxi Xiao Qingyan Bai Juntao Zhang Kecheng Zheng Xiaowei Zhou Qifeng Chen and Yujun Shen. 2023. Codef: Content deformation fields for temporally consistent video processing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.07926 (2023)."},{"key":"e_1_3_3_2_39_1","unstructured":"Jordi Pont-Tuset Federico Perazzi Sergi Caelles Pablo Arbel\u00e1ez Alex Sorkine-Hornung and Luc Van\u00a0Gool. 2017. The 2017 davis challenge on video object segmentation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1704.00675 (2017)."},{"key":"e_1_3_3_2_40_1","unstructured":"Chenyang Qi Xiaodong Cun Yong Zhang Chenyang Lei Xintao Wang Ying Shan and Qifeng Chen. 2023. FateZero: Fusing Attentions for Zero-shot Text-based Video Editing. arXiv:https:\/\/arXiv.org\/abs\/2303.09535 (2023)."},{"key":"e_1_3_3_2_41_1","first-page":"8748","volume-title":"International conference on machine learning","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong\u00a0Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et\u00a0al. 2021. Learning transferable visual models from natural language supervision. In International conference on machine learning. PMLR, 8748\u20138763."},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"publisher","unstructured":"Alex Rav-Acha Pushmeet Kohli Carsten Rother and Andrew Fitzgibbon. 2008. Unwrap mosaics: a new representation for video editing. ACM Trans. Graph. 27 3 (aug 2008) 1\u201311. 10.1145\/1360612.1360616https:\/\/dl.acm.org\/doi\/10.1145\/1360612.1360616","DOI":"10.1145\/1360612.1360616"},{"key":"e_1_3_3_2_43_1","unstructured":"Robin Rombach Andreas Blattmann Dominik Lorenz Patrick Esser and Bj\u00f6rn Ommer. 2021. High-Resolution Image Synthesis with Latent Diffusion Models. arxiv:https:\/\/arXiv.org\/abs\/2112.10752\u00a0[cs.CV]"},{"key":"e_1_3_3_2_44_1","unstructured":"Ciara Rowles. 2024. svd-temporal-controlnet. https:\/\/github.com\/CiaraStrawberry\/svd-temporal-controlnet."},{"key":"e_1_3_3_2_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.02155"},{"key":"e_1_3_3_2_46_1","doi-asserted-by":"crossref","unstructured":"Xiaoyu Shi Zhaoyang Huang Fu-Yun Wang Weikang Bian Dasong Li Yi Zhang Manyuan Zhang Ka\u00a0Chun Cheung Simon See Hongwei Qin et\u00a0al. 2024. Motion-I2V: Consistent and Controllable Image-to-Video Generation with Explicit Motion Modeling. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.15977 (2024).","DOI":"10.1145\/3641519.3657497"},{"key":"e_1_3_3_2_47_1","unstructured":"Jiaming Song Chenlin Meng and Stefano Ermon. 2020. Denoising diffusion implicit models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2010.02502 (2020)."},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00191"},{"key":"e_1_3_3_2_49_1","unstructured":"Haofan Wang Qixun Wang Xu Bai Zekui Qin and Anthony Chen. 2024b. InstantStyle: Free Lunch towards Style-Preserving in Text-to-Image Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2404.02733 (2024)."},{"key":"e_1_3_3_2_50_1","unstructured":"Jiuniu Wang Hangjie Yuan Dayou Chen Yingya Zhang Xiang Wang and Shiwei Zhang. 2023b. Modelscope text-to-video technical report. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.06571 (2023)."},{"key":"e_1_3_3_2_51_1","unstructured":"Qixun Wang Xu Bai Haofan Wang Zekui Qin and Anthony Chen. 2024a. InstantID: Zero-shot Identity-Preserving Generation in Seconds. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2401.07519 (2024)."},{"key":"e_1_3_3_2_52_1","unstructured":"Wen Wang kangyang Xie Zide Liu Hao Chen Yue Cao Xinlong Wang and Chunhua Shen. 2023a. Zero-Shot Video Editing Using Off-The-Shelf Image Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2303.17599 (2023)."},{"key":"e_1_3_3_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00701"},{"key":"e_1_3_3_2_54_1","unstructured":"Jay\u00a0Zhangjie Wu Xiuyu Li Difei Gao Zhen Dong Jinbin Bai Aishani Singh Xiaoyu Xiang Youzeng Li Zuwei Huang Yuanxi Sun et\u00a0al. 2023b. CVPR 2023 Text Guided Video Editing Competition. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.16003 (2023)."},{"key":"e_1_3_3_2_55_1","unstructured":"Guangxuan Xiao Tianwei Yin William\u00a0T. Freeman Fr\u00e9do Durand and Song Han. 2023. FastComposer: Tuning-Free Multi-Subject Image Generation with Localized Attention. arXiv (2023)."},{"key":"e_1_3_3_2_56_1","doi-asserted-by":"publisher","unstructured":"Kun Xu Yong Li Tao Ju Shi-Min Hu and Tian-Qiang Liu. 2009. Efficient affinity-based edit propagation using K-D tree. ACM Trans. Graph. 28 5 (dec 2009) 1\u20136. 10.1145\/1618452.1618464https:\/\/dl.acm.org\/doi\/10.1145\/1618452.1618464","DOI":"10.1145\/1618452.1618464"},{"key":"e_1_3_3_2_57_1","unstructured":"Xingqian Xu Jiayi Guo Zhangyang Wang Gao Huang Irfan Essa and Humphrey Shi. 2023. Prompt-Free Diffusion: Taking\" Text\" out of Text-to-Image Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.16223 (2023)."},{"key":"e_1_3_3_2_58_1","unstructured":"Hanshu Yan Jun\u00a0Hao Liew Long Mai Shanchuan Lin and Jiashi Feng. 2023b. Magicprop: Diffusion-based video editing via motion-aware appearance propagation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2309.00908 (2023)."},{"key":"e_1_3_3_2_59_1","unstructured":"Wilson Yan Andrew Brown Pieter Abbeel Rohit Girdhar and Samaneh Azadi. 2023a. Motion-Conditioned Image Animation for Video Editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2311.18827 (2023)."},{"key":"e_1_3_3_2_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3610548.3618160"},{"key":"e_1_3_3_2_61_1","unstructured":"Danah Yatim Rafail Fridman Omer Bar-Tal Yoni Kasten and Tali Dekel. 2023. Space-Time Diffusion Features for Zero-Shot Text-Driven Motion Transfer. arXiv preprint arxiv:https:\/\/arXiv.org\/abs\/2311.17009 (2023)."},{"key":"e_1_3_3_2_62_1","unstructured":"Hu Ye Jun Zhang Sibo Liu Xiao Han and Wei Yang. 2023. IP-Adapter: Text Compatible Image Prompt Adapter for Text-to-Image Diffusion Models. (2023)."},{"key":"e_1_3_3_2_63_1","unstructured":"Shengming Yin Chenfei Wu Jian Liang Jie Shi Houqiang Li Gong Ming and Nan Duan. 2023. Dragnuwa: Fine-grained control in video generation by integrating text image and trajectory. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2308.08089 (2023)."},{"key":"e_1_3_3_2_64_1","doi-asserted-by":"publisher","unstructured":"Kaan Y\u00fccer Alec Jacobson Alexander Hornung and Olga Sorkine. 2012. Transfusive image manipulation. ACM Trans. Graph. 31 6 Article 176 (nov 2012) 9\u00a0pages. 10.1145\/2366145.2366195https:\/\/dl.acm.org\/doi\/10.1145\/2366145.2366195","DOI":"10.1145\/2366145.2366195"},{"key":"e_1_3_3_2_65_1","unstructured":"Polina Zablotskaia Aliaksandr Siarohin Bo Zhao and Leonid Sigal. 2019. Dwnet: Dense warp-based network for pose-guided human video generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/1910.09139 (2019)."},{"key":"e_1_3_3_2_66_1","unstructured":"Shiwei Zhang Jiayu Wang Yingya Zhang Kang Zhao Hangjie Yuan Zhiwu Qing Xiang Wang Deli Zhao and Jingren Zhou. 2023a. I2VGen-XL: High-Quality Image-to-Video Synthesis via Cascaded Diffusion Models. (2023)."},{"key":"e_1_3_3_2_67_1","unstructured":"Yabo Zhang Yuxiang Wei Dongsheng Jiang Xiaopeng Zhang Wangmeng Zuo and Qi Tian. 2023b. ControlVideo: Training-free Controllable Text-to-Video Generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.13077 (2023)."},{"key":"e_1_3_3_2_68_1","unstructured":"Min Zhao Rongzhen Wang Fan Bao Chongxuan Li and Jun Zhu. 2023b. ControlVideo: Adding Conditional Control for One Shot Text-to-Video Editing. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2305.17098 (2023)."},{"key":"e_1_3_3_2_69_1","unstructured":"Rui Zhao Yuchao Gu Jay\u00a0Zhangjie Wu David\u00a0Junhao Zhang Jiawei Liu Weijia Wu Jussi Keppo and Mike\u00a0Zheng Shou. 2023a. MotionDirector: Motion Customization of Text-to-Video Diffusion Models. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2310.08465 (2023)."}],"event":{"name":"SA '24: SIGGRAPH Asia 2024 Conference Papers","location":"Tokyo Japan","acronym":"SA '24","sponsor":["SIGGRAPH ACM Special Interest Group on Computer Graphics and Interactive Techniques"]},"container-title":["SIGGRAPH Asia 2024 Conference Papers"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3680528.3687656","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3680528.3687656","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:18:20Z","timestamp":1750295900000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3680528.3687656"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,3]]},"references-count":68,"alternative-id":["10.1145\/3680528.3687656","10.1145\/3680528"],"URL":"https:\/\/doi.org\/10.1145\/3680528.3687656","relation":{},"subject":[],"published":{"date-parts":[[2024,12,3]]},"assertion":[{"value":"2024-12-03","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}