{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:21:10Z","timestamp":1765308070092,"version":"3.46.0"},"publisher-location":"New York, NY, USA","reference-count":84,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758140","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:50:47Z","timestamp":1761371447000},"page":"12140-12149","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Let Your Video Listen to Your Music! -- Beat-Aligned, Content-Preserving Video Editing with Arbitrary Music"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2999-3291","authenticated-orcid":false,"given":"Xinyu","family":"Zhang","sequence":"first","affiliation":[{"name":"The Australian Institute for Machine Learning, Adelaide, SA, Australia and The University of Auckland, Auckland, New Zealand"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2668-9630","authenticated-orcid":false,"given":"Dong","family":"Gong","sequence":"additional","affiliation":[{"name":"University of New South Wales, Sydney, NSW, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5681-6960","authenticated-orcid":false,"given":"Zicheng","family":"Duan","sequence":"additional","affiliation":[{"name":"The University of Adelaide, Adelaide, SA, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3027-8364","authenticated-orcid":false,"given":"Anton","family":"van den Hengel","sequence":"additional","affiliation":[{"name":"The Australian Institute for Machine Learning, Adelaide, SA, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3584-795X","authenticated-orcid":false,"given":"Lingqiao","family":"Liu","sequence":"additional","affiliation":[{"name":"The University of Adelaide, Adelaide, SA, Australia"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","first-page":"24206","article-title":"Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text","volume":"34","author":"Akbari Hassan","year":"2021","unstructured":"Hassan Akbari, Liangzhe Yuan, Rui Qian, Wei-Hong Chuang, Shih-Fu Chang, Yin Cui, and Boqing Gong. 2021. Vatt: Transformers for multimodal self-supervised learning from raw video, audio and text. NeurIPS, Vol. 34 (2021), 24206-24221.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_2_1","unstructured":"Paul Albert Frederic Z Zhang Hemanth Saratchandran Cristian Rodriguez-Opazo Anton van den Hengel and Ehsan Abbasnejad. 2025. RandLoRA: Full rank parameter-efficient fine-tuning of large models. In ICLR."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592458"},{"key":"e_1_3_2_2_4_1","doi-asserted-by":"publisher","DOI":"10.1007\/s41095-018-0115-y"},{"key":"e_1_3_2_2_5_1","unstructured":"James Betker Gabriel Goh Li Jing Tim Brooks Jianfeng Wang Linjie Li Long Ouyang Juntang Zhuang Joyce Lee Yufei Guo et al. 2023. Improving image generation with better captions. Computer Science. https:\/\/cdn.openai.com\/papers\/dall-e-3.pdf Vol. 2 3 (2023) 8."},{"key":"e_1_3_2_2_6_1","unstructured":"Andreas Blattmann Tim Dockhorn Sumith Kulal Daniel Mendelevitch Maciej Kilian Dominik Lorenz Yam Levi Zion English Vikram Voleti Adam Letts et al. 2023. Stable video diffusion: Scaling latent video diffusion models to large datasets. arXiv preprint arXiv:2311.15127 (2023)."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2007.366341"},{"key":"e_1_3_2_2_8_1","first-page":"701","article-title":"Sound2sight: Generating visual dynamics from sound and context","author":"Chatterjee Moitreya","year":"2020","unstructured":"Moitreya Chatterjee and Anoop Cherian. 2020. Sound2sight: Generating visual dynamics from sound and context. In ECCV. 701-719.","journal-title":"ECCV."},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/MMUL.2011.19"},{"volume-title":"Joint-modal label denoising for weakly-supervised audio-visual video parsing","author":"Cheng Haoyue","key":"e_1_3_2_2_10_1","unstructured":"Haoyue Cheng, Zhaoyang Liu, Hang Zhou, Chen Qian, Wayne Wu, and Limin Wang. 2022. Joint-modal label denoising for weakly-supervised audio-visual video parsing. In ECCV. Springer, 431-448."},{"key":"e_1_3_2_2_11_1","first-page":"26826","article-title":"MeLFusion: Synthesizing Music from Image and Language Cues using Diffusion Models","author":"Chowdhury Sanjoy","year":"2024","unstructured":"Sanjoy Chowdhury, Sayan Nag, KJ Joseph, Balaji Vasan Srinivasan, and Dinesh Manocha. 2024. MeLFusion: Synthesizing Music from Image and Language Cues using Diffusion Models. In CVPR. 26826-26835.","journal-title":"CVPR."},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-54427-4_19"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201371"},{"key":"e_1_3_2_2_14_1","unstructured":"Qixin Deng Qikai Yang Ruibin Yuan Yipeng Huang Yi Wang Xubo Liu Zeyue Tian Jiahao Pan Ge Zhang Hanfeng Lin et al. 2024. ComposerX: Multi-Agent Symbolic Music Composition with LLMs. arXiv preprint arXiv:2404.18081 (2024)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1080\/09298210701653344"},{"key":"e_1_3_2_2_16_1","first-page":"12873","article-title":"Taming transformers for high-resolution image synthesis","author":"Esser Patrick","year":"2021","unstructured":"Patrick Esser, Robin Rombach, and Bjorn Ommer. 2021. Taming transformers for high-resolution image synthesis. In CVPR. 12873-12883.","journal-title":"CVPR."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/3001773.3001782"},{"key":"e_1_3_2_2_18_1","first-page":"758","article-title":"Foley music: Learning to generate music from videos","author":"Gan Chuang","year":"2020","unstructured":"Chuang Gan, Deng Huang, Peihao Chen, Joshua B Tenenbaum, and Antonio Torralba. 2020. Foley music: Learning to generate music from videos. In ECCV. 758-775.","journal-title":"ECCV."},{"volume-title":"Long video generation with time-agnostic vqgan and time-sensitive transformer","author":"Ge Songwei","key":"e_1_3_2_2_19_1","unstructured":"Songwei Ge, Thomas Hayes, Harry Yang, Xi Yin, Guan Pang, David Jacobs, Jia-Bin Huang, and Devi Parikh. 2022. Long video generation with time-agnostic vqgan and time-sensitive transformer. In ECCV. Springer, 102-118."},{"key":"e_1_3_2_2_20_1","first-page":"15180","article-title":"Imagebind: One embedding space to bind them all","author":"Girdhar Rohit","year":"2023","unstructured":"Rohit Girdhar, Alaaeldin El-Nouby, Zhuang Liu, Mannat Singh, Kalyan Vasudev Alwala, Armand Joulin, and Ishan Misra. 2023. Imagebind: One embedding space to bind them all. In CVPR. 15180-15190.","journal-title":"CVPR."},{"key":"e_1_3_2_2_21_1","volume-title":"Contrastive audio-visual masked autoencoder. arXiv preprint arXiv:2210.07839","author":"Gong Yuan","year":"2022","unstructured":"Yuan Gong, Andrew Rouditchenko, Alexander H Liu, David Harwath, Leonid Karlinsky, Hilde Kuehne, and James Glass. 2022. Contrastive audio-visual masked autoencoder. arXiv preprint arXiv:2210.07839 (2022)."},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682863"},{"key":"e_1_3_2_2_23_1","first-page":"6840","article-title":"Denoising diffusion probabilistic models","volume":"33","author":"Ho Jonathan","year":"2020","unstructured":"Jonathan Ho, Ajay Jain, and Pieter Abbeel. 2020. Denoising diffusion probabilistic models. NeurIPS, Vol. 33 (2020), 6840-6851.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_24_1","unstructured":"Jonathan Ho and Tim Salimans. 2022. Classifier-Free Diffusion Guidance. showeprintarXiv:2207.12598"},{"key":"e_1_3_2_2_25_1","first-page":"8633","article-title":"Video diffusion models","volume":"35","author":"Ho Jonathan","year":"2022","unstructured":"Jonathan Ho, Tim Salimans, Alexey Gritsenko, William Chan, Mohammad Norouzi, and David J Fleet. 2022. Video diffusion models. NeurIPS, Vol. 35 (2022), 8633-8646.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_26_1","volume-title":"The Eleventh International Conference on Learning Representations.","author":"Hong Wenyi","year":"2023","unstructured":"Wenyi Hong, Ming Ding, Wendi Zheng, Xinghan Liu, and Jie Tang. 2023. CogVideo: Large-scale Pretraining for Text-to-Video Generation via Transformers. In The Eleventh International Conference on Learning Representations."},{"key":"e_1_3_2_2_27_1","unstructured":"Edward J Hu Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang Weizhu Chen et al. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In ICLR."},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1145\/1027527.1027641"},{"key":"e_1_3_2_2_29_1","volume-title":"Multi-modal Music Understanding and Generation with the Power of Large Language Models. arXiv preprint arXiv:2311.11255","author":"Hussain Atin Sakkeer","year":"2023","unstructured":"Atin Sakkeer Hussain, Shansong Liu, Chenshuo Sun, and Ying Shan. 2023. M^2UGen: Multi-modal Music Understanding and Generation with the Power of Large Language Models. arXiv preprint arXiv:2311.11255 (2023)."},{"key":"e_1_3_2_2_30_1","volume-title":"Video2Music: Suitable Music Generation from Videos using an Affective Multimodal Transformer model. arXiv preprint arXiv:2311.00968","author":"Kang Jaeyong","year":"2023","unstructured":"Jaeyong Kang, Soujanya Poria, and Dorien Herremans. 2023. Video2Music: Suitable Music Generation from Videos using an Affective Multimodal Transformer model. arXiv preprint arXiv:2311.00968 (2023)."},{"key":"e_1_3_2_2_31_1","first-page":"13401","article-title":"Ai choreographer: Music conditioned 3d dance generation with aist","author":"Li Ruilong","year":"2021","unstructured":"Ruilong Li, Shan Yang, David A Ross, and Angjoo Kanazawa. 2021. Ai choreographer: Music conditioned 3d dance generation with aist. In ICCV. 13401-13412.","journal-title":"ICCV."},{"key":"e_1_3_2_2_32_1","volume-title":"VidMusician: Video-to-Music Generation with Semantic-Rhythmic Alignment via Hierarchical Visual Features. arXiv preprint arXiv:2412.06296","author":"Li Sifei","year":"2024","unstructured":"Sifei Li, Binxin Yang, Chunji Yin, Chong Sun, Yuxin Zhang, Weiming Dong, and Chen Li. 2024. VidMusician: Video-to-Music Generation with Semantic-Rhythmic Alignment via Hierarchical Visual Features. arXiv preprint arXiv:2412.06296 (2024)."},{"key":"e_1_3_2_2_33_1","first-page":"109790","article-title":"Evaluation of text-to-video generation models: A dynamics perspective","volume":"37","author":"Liao Mingxiang","year":"2024","unstructured":"Mingxiang Liao, Qixiang Ye, Wangmeng Zuo, Fang Wan, Tianyu Wang, Yuzhong Zhao, Jingdong Wang, Xinyu Zhang, et al., 2024. Evaluation of text-to-video generation models: A dynamics perspective. NeurIPS, Vol. 37 (2024), 109790-109816.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/2766966","article-title":"Audeosynth: music-driven video montage","volume":"34","author":"Liao Zicheng","year":"2015","unstructured":"Zicheng Liao, Yizhou Yu, Bingchen Gong, and Lechao Cheng. 2015. Audeosynth: music-driven video montage. ACM Transactions on Graphics (TOG), Vol. 34, 4 (2015), 1-10.","journal-title":"ACM Transactions on Graphics (TOG)"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1145\/2733373.2806359"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1145\/3123266.3123399"},{"key":"e_1_3_2_2_37_1","volume-title":"MotionClone: Training-Free Motion Cloning for Controllable Video Generation. ICLR","author":"Ling Pengyang","year":"2025","unstructured":"Pengyang Ling, Jiazi Bu, Pan Zhang, Xiaoyi Dong, Yuhang Zang, Tong Wu, Huaian Chen, Jiaqi Wang, and Yi Jin. 2025. MotionClone: Training-Free Motion Cloning for Controllable Video Generation. ICLR (2025)."},{"key":"e_1_3_2_2_38_1","volume-title":"Unconditional audio generation with generative adversarial networks and cycle regularization. arXiv preprint arXiv:2005.08526","author":"Liu Jen-Yu","year":"2020","unstructured":"Jen-Yu Liu, Yu-Hua Chen, Yin-Cheng Yeh, and Yi-Hsuan Yang. 2020. Unconditional audio generation with generative adversarial networks and cycle regularization. arXiv preprint arXiv:2005.08526 (2020)."},{"key":"e_1_3_2_2_39_1","volume-title":"Kwang-Ting Cheng, and Min-Hung Chen.","author":"Liu Shih-Yang","year":"2024","unstructured":"Shih-Yang Liu, Chien-Yi Wang, Hongxu Yin, Pavlo Molchanov, Yu-Chiang Frank Wang, Kwang-Ting Cheng, and Min-Hung Chen. 2024. Dora: Weight-decomposed low-rank adaptation. In ICML."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11390-023-3064-6"},{"key":"e_1_3_2_2_41_1","unstructured":"Yu Lu Yuanzhi Liang Linchao Zhu and Yi Yang. 2024. FreeLong: Training-Free Long Video Generation with SpectralBlend Temporal Attention. In NeurIPS."},{"key":"e_1_3_2_2_42_1","volume-title":"Msanii: High Fidelity Music Synthesis on a Shoestring Budget. arXiv preprint arXiv:2301.06468","author":"Maina Kinyugo","year":"2023","unstructured":"Kinyugo Maina. 2023. Msanii: High Fidelity Music Synthesis on a Shoestring Budget. arXiv preprint arXiv:2301.06468 (2023)."},{"key":"e_1_3_2_2_43_1","volume-title":"Symbolic music generation with diffusion models. arXiv preprint arXiv:2103.16091","author":"Mittal Gautam","year":"2021","unstructured":"Gautam Mittal, Jesse Engel, Curtis Hawthorne, and Ian Simon. 2021. Symbolic music generation with diffusion models. arXiv preprint arXiv:2103.16091 (2021)."},{"key":"e_1_3_2_2_44_1","volume-title":"Openvid-1m: A large-scale high-quality dataset for text-to-video generation. arXiv preprint arXiv:2407.02371","author":"Nan Kepan","year":"2024","unstructured":"Kepan Nan, Rui Xie, Penghao Zhou, Tiehan Fan, Zhenheng Yang, Zhijie Chen, Xiang Li, Jian Yang, and Ying Tai. 2024. Openvid-1m: A large-scale high-quality dataset for text-to-video generation. arXiv preprint arXiv:2407.02371 (2024)."},{"key":"e_1_3_2_2_45_1","volume-title":"GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. showeprintarXiv:2112.10741","author":"Nichol Alex","year":"2021","unstructured":"Alex Nichol, Prafulla Dhariwal, Aditya Ramesh, Pranav Shyam, Pamela Mishkin, Bob McGrew, Ilya Sutskever, and Mark Chen. 2021. GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models. showeprintarXiv:2112.10741"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2011.2181492"},{"key":"e_1_3_2_2_47_1","unstructured":"OpenAI. 2024. Video generation models as world simulators. Technical Report. OpenAI. https:\/\/openai.com\/research\/video-generation-models-as-world-simulators"},{"key":"e_1_3_2_2_48_1","first-page":"8748","article-title":"Learning transferable visual models from natural language supervision","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, et al., 2021. Learning transferable visual models from natural language supervision. In ICML. PMLR, 8748-8763.","journal-title":"ICML. PMLR"},{"key":"e_1_3_2_2_49_1","volume-title":"Avlnet: Learning audio-visual language representations from instructional videos. arXiv preprint arXiv:2006.09199","author":"Rouditchenko Andrew","year":"2020","unstructured":"Andrew Rouditchenko, Angie Boggust, David Harwath, Brian Chen, Dhiraj Joshi, Samuel Thomas, Kartik Audhkhasi, Hilde Kuehne, Rameswar Panda, Rogerio Feris, et al., 2020. Avlnet: Learning audio-visual language representations from instructional videos. arXiv preprint arXiv:2006.09199 (2020)."},{"key":"e_1_3_2_2_50_1","first-page":"10219","article-title":"Mm-diffusion: Learning multi-modal diffusion models for joint audio and video generation","author":"Ruan Ludan","year":"2023","unstructured":"Ludan Ruan, Yiyang Ma, Huan Yang, Huiguo He, Bei Liu, Jianlong Fu, Nicholas Jing Yuan, Qin Jin, and Baining Guo. 2023. Mm-diffusion: Learning multi-modal diffusion models for joint audio and video generation. In CVPR. 10219-10228.","journal-title":"CVPR."},{"key":"e_1_3_2_2_51_1","first-page":"36479","article-title":"Photorealistic text-to-image diffusion models with deep language understanding","volume":"35","author":"Saharia Chitwan","year":"2022","unstructured":"Chitwan Saharia, William Chan, Saurabh Saxena, Lala Li, Jay Whang, Emily L Denton, Kamyar Ghasemipour, Raphael Gontijo Lopes, Burcu Karagol Ayan, Tim Salimans, et al., 2022. Photorealistic text-to-image diffusion models with deep language understanding. NeurIPS, Vol. 35 (2022), 36479-36494.","journal-title":"NeurIPS"},{"key":"e_1_3_2_2_52_1","volume-title":"Mousai: Text-to-music generation with long-context latent diffusion. arXiv preprint arXiv:2301.11757","author":"Schneider Flavio","year":"2023","unstructured":"Flavio Schneider, Ojasv Kamal, Zhijing Jin, and Bernhard Sch\u00f6lkopf. 2023. Mousai: Text-to-music generation with long-context latent diffusion. arXiv preprint arXiv:2301.11757 (2023)."},{"key":"e_1_3_2_2_53_1","volume-title":"Hammond","author":"Shamma David A.","year":"2005","unstructured":"David A. Shamma, Bryan Pardo, and Kristian J. Hammond. 2005. MusicStory: a personalized music video creator. In ACMMM."},{"key":"e_1_3_2_2_54_1","unstructured":"Bowen Shi Wei-Ning Hsu Kushal Lakhotia and Abdelrahman Mohamed. 2022. Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction. In ICLR."},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/SMC.2016.7844629"},{"key":"e_1_3_2_2_56_1","unstructured":"Uriel Singer Adam Polyak Thomas Hayes Xi Yin Jie An Songyang Zhang Qiyuan Hu Harry Yang Oron Ashual Oran Gafni et al. 2023. Make-A-Video: Text-to-Video Generation without Text-Video Data. In ICLR."},{"key":"e_1_3_2_2_57_1","volume-title":"Denoising Diffusion Implicit Models. arXiv:2010.02502 (October","author":"Song Jiaming","year":"2020","unstructured":"Jiaming Song, Chenlin Meng, and Stefano Ermon. 2020. Denoising Diffusion Implicit Models. arXiv:2010.02502 (October 2020)."},{"key":"e_1_3_2_2_58_1","unstructured":"Yang Song Jascha Sohl-Dickstein Diederik P Kingma Abhishek Kumar Stefano Ermon and Ben Poole. 2021. Score-Based Generative Modeling through Stochastic Differential Equations. In ICLR."},{"key":"e_1_3_2_2_59_1","volume-title":"Qingqing Huang, Dima Kuzmin, Joonseok Lee, Chris Donahue, Fei Sha, Aren Jansen, Yu Wang, Mauro Verzetti, et al.","author":"Su Kun","year":"2023","unstructured":"Kun Su, Judith Yue Li, Qingqing Huang, Dima Kuzmin, Joonseok Lee, Chris Donahue, Fei Sha, Aren Jansen, Yu Wang, Mauro Verzetti, et al., 2023. V2Meow: Meowing to the Visual Beat via Music Generation. arXiv preprint arXiv:2305.06594 (2023)."},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1145\/3592118"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"crossref","unstructured":"Kim Sung-Bin Arda Senocak Hyunwoo Ha Andrew Owens and Tae-Hyun Oh. 2023. Sound to visual scene generation by audio-to-visual latent alignment. In sung2023sound. 6430-6440.","DOI":"10.1109\/CVPR52729.2023.00622"},{"key":"e_1_3_2_2_62_1","volume-title":"Vidmuse: A simple video-to-music generation framework with long-short-term modeling. arXiv preprint arXiv:2406.04321","author":"Tian Zeyue","year":"2024","unstructured":"Zeyue Tian, Zhaoyang Liu, Ruibin Yuan, Jiahao Pan, Qifeng Liu, Xu Tan, Qifeng Chen, Wei Xue, and Yike Guo. 2024. Vidmuse: A simple video-to-music generation framework with long-short-term modeling. arXiv preprint arXiv:2406.04321 (2024)."},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"crossref","unstructured":"Pauli Virtanen Ralf Gommers Travis E Oliphant Matt Haberland Tyler Reddy David Cournapeau Evgeni Burovski Pearu Peterson Warren Weckesser Jonathan Bright et al. 2020. SciPy 1.0: fundamental algorithms for scientific computing in Python. Nature methods Vol. 17 3 (2020) 261-272.","DOI":"10.1038\/s41592-020-0772-5"},{"key":"e_1_3_2_2_64_1","volume-title":"Wan: Open and Advanced Large-Scale Video Generative Models. arXiv preprint arXiv:2503.20314","author":"Wan Team","year":"2025","unstructured":"Team Wan, Ang Wang, Baole Ai, Bin Wen, Chaojie Mao, Chen-Wei Xie, Di Chen, Feiwu Yu, Haiming Zhao, Jianxiao Yang, Jianyuan Zeng, Jiayu Wang, Jingfeng Zhang, Jingren Zhou, Jinkai Wang, Jixuan Chen, Kai Zhu, Kang Zhao, Keyu Yan, Lianghua Huang, Mengyang Feng, Ningyi Zhang, Pandeng Li, Pingyu Wu, Ruihang Chu, Ruili Feng, Shiwei Zhang, Siyang Sun, Tao Fang, Tianxing Wang, Tianyi Gui, Tingyu Weng, Tong Shen, Wei Lin, Wei Wang, Wei Wang, Wenmeng Zhou, Wente Wang, Wenting Shen, Wenyuan Yu, Xianzhong Shi, Xiaoming Huang, Xin Xu, Yan Kou, Yangyu Lv, Yifei Li, Yijing Liu, Yiming Wang, Yingya Zhang, Yitong Huang, Yong Li, You Wu, Yu Liu, Yulin Pan, Yun Zheng, Yuntao Hong, Yupeng Shi, Yutong Feng, Zeyinzi Jiang, Zhen Han, Zhi-Fan Wu, and Ziyu Liu. 2025. Wan: Open and Advanced Large-Scale Video Generative Models. arXiv preprint arXiv:2503.20314 (2025)."},{"key":"e_1_3_2_2_65_1","first-page":"10087","article-title":"Self-expansion of pre-trained models with mixture of adapters for continual learning","author":"Wang Huiyi","year":"2025","unstructured":"Huiyi Wang, Haodong Lu, Lina Yao, and Dong Gong. 2025a. Self-expansion of pre-trained models with mixture of adapters for continual learning. In CVPR. 10087-10098.","journal-title":"CVPR."},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"crossref","unstructured":"Jianren Wang Zhaoyuan Fang and Hang Zhao. 2020. AlignNet: A Unifying Approach to Audio-Visual Alignment. In WACV.","DOI":"10.1109\/WACV45572.2020.9093345"},{"key":"e_1_3_2_2_67_1","volume-title":"AV-DiT: Efficient Audio-Visual Diffusion Transformer for Joint Audio and Video Generation. arXiv preprint arXiv:2406.07686","author":"Wang Kai","year":"2024","unstructured":"Kai Wang, Shijian Deng, Jing Shi, Dimitrios Hatzinakos, and Yapeng Tian. 2024. AV-DiT: Efficient Audio-Visual Diffusion Transformer for Joint Audio and Video Generation. arXiv preprint arXiv:2406.07686 (2024)."},{"key":"e_1_3_2_2_68_1","volume-title":"Framer: Interactive Frame Interpolation. In ICLR.","author":"Wang Wen","year":"2025","unstructured":"Wen Wang, Qiuyu Wang, Kecheng Zheng, Hao OUYANG, Zhekai Chen, Biao Gong, Hao Chen, Yujun Shen, and Chunhua Shen. 2025b. Framer: Interactive Frame Interpolation. In ICLR."},{"key":"e_1_3_2_2_69_1","volume-title":"VideoUFO: A Million-Scale User-Focused Dataset for Text-to-Video Generation. arXiv preprint arXiv:2503.01739","author":"Wang Wenhao","year":"2025","unstructured":"Wenhao Wang and Yi Yang. 2025. VideoUFO: A Million-Scale User-Focused Dataset for Text-to-Video Generation. arXiv preprint arXiv:2503.01739 (2025)."},{"key":"e_1_3_2_2_70_1","volume-title":"Next-gpt: Any-to-any multimodal llm. arXiv preprint arXiv:2309.05519","author":"Wu Shengqiong","year":"2023","unstructured":"Shengqiong Wu, Hao Fei, Leigang Qu, Wei Ji, and Tat-Seng Chua. 2023. Next-gpt: Any-to-any multimodal llm. arXiv preprint arXiv:2309.05519 (2023)."},{"key":"e_1_3_2_2_71_1","first-page":"28984","article-title":"Moviebench: A hierarchical movie level dataset for long video generation","author":"Wu Weijia","year":"2025","unstructured":"Weijia Wu, Mingyu Liu, Zeyu Zhu, Xi Xia, Haoen Feng, Wen Wang, Kevin Qinghong Lin, Chunhua Shen, and Mike Zheng Shou. 2025. Moviebench: A hierarchical movie level dataset for long video generation. In CVPR. 28984-28994.","journal-title":"CVPR."},{"key":"e_1_3_2_2_72_1","first-page":"13220","article-title":"DyMO","author":"Xie Xin","year":"2025","unstructured":"Xin Xie and Dong Gong. 2025. DyMO: Training-Free Diffusion Model Alignment with Dynamic Multi-Objective Scheduling. In CVPR. 13220-13230.","journal-title":"Training-Free Diffusion Model Alignment with Dynamic Multi-Objective Scheduling. In CVPR."},{"key":"e_1_3_2_2_73_1","volume-title":"Seeing and Hearing: Open-domain Visual-Audio Generation with Diffusion Latent Aligners. arXiv preprint arXiv:2402.17723","author":"Xing Yazhou","year":"2024","unstructured":"Yazhou Xing, Yingqing He, Zeyue Tian, Xintao Wang, and Qifeng Chen. 2024. Seeing and Hearing: Open-domain Visual-Audio Generation with Diffusion Latent Aligners. arXiv preprint arXiv:2402.17723 (2024)."},{"key":"e_1_3_2_2_74_1","volume-title":"Diffsound: Discrete diffusion model for text-to-sound generation","author":"Yang Dongchao","year":"2023","unstructured":"Dongchao Yang, Jianwei Yu, Helin Wang, Wen Wang, Chao Weng, Yuexian Zou, and Dong Yu. 2023. Diffsound: Discrete diffusion model for text-to-sound generation. IEEE\/ACM Transactions on Audio, Speech, and Language Processing (2023)."},{"key":"e_1_3_2_2_75_1","volume-title":"Cogvideox: Text-to-video diffusion models with an expert transformer. arXiv preprint arXiv:2408.06072","author":"Yang Zhuoyi","year":"2024","unstructured":"Zhuoyi Yang, Jiayan Teng, Wendi Zheng, Ming Ding, Shiyu Huang, Jiazheng Xu, Yuanming Yang, Wenyi Hong, Xiaohan Zhang, Guanyu Feng, et al., 2024. Cogvideox: Text-to-video diffusion models with an expert transformer. arXiv preprint arXiv:2408.06072 (2024)."},{"key":"e_1_3_2_2_76_1","volume-title":"Self-supervised learning of music-dance representation through explicit-implicit rhythm synchronization. arXiv preprint arXiv:2207.03190","author":"Yu Jiashuo","year":"2022","unstructured":"Jiashuo Yu, Junfu Pu, Ying Cheng, Rui Feng, and Ying Shan. 2022. Self-supervised learning of music-dance representation through explicit-implicit rhythm synchronization. arXiv preprint arXiv:2207.03190 (2022)."},{"key":"e_1_3_2_2_77_1","doi-asserted-by":"publisher","DOI":"10.1145\/3512527.3531430"},{"key":"e_1_3_2_2_78_1","unstructured":"Qingru Zhang Minshuo Chen Alexander Bukharin Pengcheng He Yu Cheng Weizhu Chen and Tuo Zhao. 2023. Adaptive Budget Allocation for Parameter-Efficient Fine-Tuning. In ICLR."},{"key":"e_1_3_2_2_79_1","first-page":"586","article-title":"The unreasonable effectiveness of deep features as a perceptual metric","author":"Zhang Richard","year":"2018","unstructured":"Richard Zhang, Phillip Isola, Alexei A Efros, Eli Shechtman, and Oliver Wang. 2018. The unreasonable effectiveness of deep features as a perceptual metric. In CVPR. 586-595.","journal-title":"CVPR."},{"key":"e_1_3_2_2_80_1","first-page":"5048","article-title":"AutoLoRA","author":"Zhang Ruiyi","year":"2024","unstructured":"Ruiyi Zhang, Rushi Qiang, Sai Ashish Somayajula, and Pengtao Xie. 2024. AutoLoRA: Automatically Tuning Matrix Ranks in Low-Rank Adaptation Based on Meta Learning. In NAACL. 5048-5060.","journal-title":"In NAACL."},{"key":"e_1_3_2_2_81_1","volume-title":"Training-free motion-guided video generation with enhanced temporal consistency using motion consistency loss. arXiv preprint arXiv:2501.07563","author":"Zhang Xinyu","year":"2025","unstructured":"Xinyu Zhang, Zicheng Duan, Dong Gong, and Lingqiao Liu. 2025. Training-free motion-guided video generation with enhanced temporal consistency using motion consistency loss. arXiv preprint arXiv:2501.07563 (2025)."},{"key":"e_1_3_2_2_82_1","unstructured":"Min Zhao Guande He Yixiao Chen Hongzhou Zhu Chongxuan Li and Jun Zhu. 2025. RIFLEx: A Free Lunch for Length Extrapolation in Video Diffusion Transformers. In ICML."},{"key":"e_1_3_2_2_83_1","doi-asserted-by":"publisher","DOI":"10.1007\/s41095-022-0292-6"},{"key":"e_1_3_2_2_84_1","doi-asserted-by":"publisher","DOI":"10.1145\/3485664"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","sponsor":["SIGMM ACM Special Interest Group on Multimedia"],"location":"Dublin Ireland","acronym":"MM '25"},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758140","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:19:08Z","timestamp":1765307948000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758140"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":84,"alternative-id":["10.1145\/3746027.3758140","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758140","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}