{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:47:41Z","timestamp":1776887261814,"version":"3.51.2"},"publisher-location":"New York, NY, USA","reference-count":62,"publisher":"ACM","content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3755831","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:56:43Z","timestamp":1761371803000},"page":"10671-10680","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["DualDub: Video-to-Soundtrack Generation via Joint Speech and Background Audio Synthesis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-3859-0927","authenticated-orcid":false,"given":"Wenjie","family":"Tian","sequence":"first","affiliation":[{"name":"Northwestern Polytechnical University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9275-523X","authenticated-orcid":false,"given":"Xinfa","family":"Zhu","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1036-7888","authenticated-orcid":false,"given":"Haohe","family":"Liu","sequence":"additional","affiliation":[{"name":"University of Surrey, Guildford, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1136-8279","authenticated-orcid":false,"given":"Zhixian","family":"Zhao","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0005-5413-6725","authenticated-orcid":false,"given":"Zihao","family":"Chen","sequence":"additional","affiliation":[{"name":"Giant Network Inc., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-2051-4538","authenticated-orcid":false,"given":"Chaofan","family":"Ding","sequence":"additional","affiliation":[{"name":"Giant Network Inc., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8855-8628","authenticated-orcid":false,"given":"Xinhan","family":"Di","sequence":"additional","affiliation":[{"name":"Giant Network Inc., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0003-2602-2910","authenticated-orcid":false,"given":"Junjie","family":"Zheng","sequence":"additional","affiliation":[{"name":"Giant Network Inc., Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5406-2376","authenticated-orcid":false,"given":"Lei","family":"Xie","sequence":"additional","affiliation":[{"name":"Northwestern Polytechnical University, Xi'an, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447911"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053174"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.02056"},{"key":"e_1_3_2_1_4_1","volume-title":"BEATs: Audio Pre-Training with Acoustic Tokenizers. In International Conference on Machine Learning, ICML 2023","volume":"5193","author":"Chen Sanyuan","year":"2023","unstructured":"Sanyuan Chen, Yu Wu, Chengyi Wang, Shujie Liu, Daniel Tompkins, Zhuo Chen, Wanxiang Che, Xiangzhan Yu, and Furu Wei. 2023. BEATs: Audio Pre-Training with Acoustic Tokenizers. In International Conference on Machine Learning, ICML 2023, 23-29 July 2023, Honolulu, Hawaii, USA (Proceedings of Machine Learning Research, Vol. 202), Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett (Eds.). PMLR, 5178-5193."},{"key":"e_1_3_2_1_5_1","volume-title":"F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching. CoRR","author":"Chen Yushen","year":"2024","unstructured":"Yushen Chen, Zhikang Niu, Ziyang Ma, Keqi Deng, Chunhui Wang, Jian Zhao, Kai Yu, and Xie Chen. 2024. F5-TTS: A Fairytaler that Fakes Fluent and Faithful Speech with Flow Matching. CoRR, Vol. abs\/2410.06885 (2024)."},{"key":"e_1_3_2_1_6_1","first-page":"6147","volume-title":"Large-Scale Self-Supervised Speech Representation Learning for Automatic Speaker Verification. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2022","author":"Chen Zhengyang","year":"2022","unstructured":"Zhengyang Chen, Sanyuan Chen, Yu Wu, Yao Qian, Chengyi Wang, Shujie Liu, Yanmin Qian, and Michael Zeng. 2022a. Large-Scale Self-Supervised Speech Representation Learning for Automatic Speaker Verification. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2022, Virtual and Singapore, 23-27 May 2022. IEEE, 6147-6151."},{"key":"e_1_3_2_1_7_1","volume-title":"Video-Guided Foley Sound Generation with Multimodal Controls. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2025","author":"Chen Ziyang","year":"2025","unstructured":"Ziyang Chen, Prem Seetharaman, Bryan C. Russell, Oriol Nieto, David Bourgin, Andrew Owens, and Justin Salamon. 2025. Video-Guided Foley Sound Generation with Multimodal Controls. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2025, Nashville, TN, USA, June 11-15, 2025. Computer Vision Foundation \/ IEEE, 18770-18781."},{"key":"e_1_3_2_1_8_1","volume-title":"Taming Multimodal Joint Training for High-Quality Video-to-Audio Synthesis. CoRR","author":"Cheng Ho Kei","year":"2024","unstructured":"Ho Kei Cheng, Masato Ishii, Akio Hayakawa, Takashi Shibuya, Alexander G. Schwing, and Yuki Mitsufuji. 2024. Taming Multimodal Joint Training for High-Quality Video-to-Audio Synthesis. CoRR, Vol. abs\/2412.15322 (2024)."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV51070.2023.00718"},{"key":"e_1_3_2_1_10_1","volume-title":"Joon Son Chung, and Shujie Liu","author":"Choi Jeongsoo","year":"2024","unstructured":"Jeongsoo Choi, Ji-Hoon Kim, Jinyu Li, Joon Son Chung, and Shujie Liu. 2024. V2SFlow: Video-to-Speech Generation with Speech Decomposition and Rectified Flow. CoRR, Vol. abs\/2411.19486 (2024)."},{"key":"e_1_3_2_1_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01411"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.findings-acl.404"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.00240"},{"key":"e_1_3_2_1_14_1","volume-title":"CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models. CoRR","author":"Du Zhihao","year":"2024","unstructured":"Zhihao Du, Yuxuan Wang, Qian Chen, Xian Shi, Xiang Lv, Tianyu Zhao, Zhifu Gao, Yexin Yang, Changfeng Gao, Hui Wang, Fan Yu, Huadai Liu, Zhengyan Sheng, Yue Gu, Chong Deng, Wen Wang, Shiliang Zhang, Zhijie Yan, and Jingren Zhou. 2024. CosyVoice 2: Scalable Streaming Speech Synthesis with Large Language Models. CoRR, Vol. abs\/2412.10117 (2024)."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"e_1_3_2_1_16_1","volume-title":"Forty-first international conference on machine learning.","author":"Esser Patrick","unstructured":"Patrick Esser, Sumith Kulal, Andreas Blattmann, Rahim Entezari, Jonas M\u00fcller, Harry Saini, Yam Levi, Dominik Lorenz, Axel Sauer, Frederic Boesel, et al., [n.d.]. Scaling rectified flow transformers for high-resolution image synthesis. In Forty-first international conference on machine learning."},{"key":"e_1_3_2_1_17_1","volume-title":"Fast Timing-Conditioned Latent Audio Diffusion. In Forty-first International Conference on Machine Learning, ICML 2024","author":"Evans Zach","year":"2024","unstructured":"Zach Evans, CJ Carr, Josiah Taylor, Scott H. Hawley, and Jordi Pons. 2024a. Fast Timing-Conditioned Latent Audio Diffusion. In Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024. OpenReview.net."},{"key":"e_1_3_2_1_18_1","first-page":"429","volume-title":"Proceedings of the 25th International Society for Music Information Retrieval Conference, ISMIR 2024","author":"Evans Zach","year":"2024","unstructured":"Zach Evans, Julian D. Parker, CJ Carr, Zachary Zukowski, Josiah Taylor, and Jordi Pons. 2024b. Long-Form Music Generation With Latent Diffusion. In Proceedings of the 25th International Society for Music Information Retrieval Conference, ISMIR 2024, San Francisco, California, USA and Online, November 10-14, 2024, Blair Kaneshiro, Gautham J. Mysore, Oriol Nieto, Chris Donahue, Cheng-Zhi Anna Huang, Jin Ha Lee, Brian McFee, and Matthew C. McCallum (Eds.). 429-437."},{"key":"e_1_3_2_1_19_1","volume-title":"MINT: a Multi-modal Image and Narrative Text Dubbing Dataset for Foley Audio Content Planning and Generation. CoRR","author":"Fu Ruibo","year":"2024","unstructured":"Ruibo Fu, Shuchen Shi, Hongming Guo, Tao Wang, Chunyu Qiang, Zhengqi Wen, Jianhua Tao, Xin Qi, Yi Lu, Xiaopeng Wang, Zhiyong Wang, Yukun Liu, Xuefei Liu, Shuai Zhang, and Guanjun Li. 2024. MINT: a Multi-modal Image and Narrative Text Dubbing Dataset for Foley Audio Content Planning and Generation. CoRR, Vol. abs\/2406.10591 (2024)."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01457"},{"key":"e_1_3_2_1_21_1","volume-title":"Contrastive audio-visual masked autoencoder. arXiv preprint arXiv:2210.07839","author":"Gong Yuan","year":"2022","unstructured":"Yuan Gong, Andrew Rouditchenko, Alexander H Liu, David Harwath, Leonid Karlinsky, Hilde Kuehne, and James Glass. 2022. Contrastive audio-visual masked autoencoder. arXiv preprint arXiv:2210.07839 (2022)."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01033"},{"key":"e_1_3_2_1_23_1","first-page":"131","article-title":"CNN architectures for large-scale audio classification. In 2017 ieee international conference on acoustics, speech and signal processing (icassp)","author":"Hershey Shawn","year":"2017","unstructured":"Shawn Hershey, Sourish Chaudhuri, Daniel PW Ellis, Jort F Gemmeke, Aren Jansen, R Channing Moore, Manoj Plakal, Devin Platt, Rif A Saurous, Bryan Seybold, et al., 2017. CNN architectures for large-scale audio classification. In 2017 ieee international conference on acoustics, speech and signal processing (icassp). IEEE, 131-135.","journal-title":"IEEE"},{"key":"e_1_3_2_1_24_1","volume-title":"Music, Sound, and Talking Head. In Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI","author":"Huang Rongjie","year":"2024","unstructured":"Rongjie Huang, Mingze Li, Dongchao Yang, Jiatong Shi, Xuankai Chang, Zhenhui Ye, Yuning Wu, Zhiqing Hong, Jiawei Huang, Jinglin Liu, Yi Ren, Yuexian Zou, Zhou Zhao, and Shinji Watanabe. 2024. AudioGPT: Understanding and Generating Speech, Music, Sound, and Talking Head. In Thirty-Eighth AAAI Conference on Artificial Intelligence, AAAI 2024, Thirty-Sixth Conference on Innovative Applications of Artificial Intelligence, IAAI 2024, Fourteenth Symposium on Educational Advances in Artificial Intelligence, EAAI 2014, February 20-27, 2024, Vancouver, Canada, Michael J. Wooldridge, Jennifer G. Dy, and Sriraam Natarajan (Eds.). AAAI Press, 23802-23804."},{"key":"e_1_3_2_1_25_1","volume-title":"Taming Visually Guided Sound Generation. In British Machine Vision Conference.","author":"Iashin Vladimir","year":"2021","unstructured":"Vladimir Iashin and Esa Rahtu. 2021. Taming Visually Guided Sound Generation. In British Machine Vision Conference."},{"key":"e_1_3_2_1_26_1","first-page":"5325","volume-title":"Synchformer: Efficient Synchronization From Sparse Cues. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2024","author":"Iashin Vladimir","year":"2024","unstructured":"Vladimir Iashin, Weidi Xie, Esa Rahtu, and Andrew Zisserman. 2024. Synchformer: Efficient Synchronization From Sparse Cues. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2024, Seoul, Republic of Korea, April 14-19, 2024. IEEE, 5325-5329."},{"key":"e_1_3_2_1_27_1","volume-title":"WavTokenizer: an Efficient Acoustic Discrete Codec Tokenizer for Audio Language Modeling. CoRR","author":"Ji Shengpeng","year":"2024","unstructured":"Shengpeng Ji, Ziyue Jiang, Xize Cheng, Yifu Chen, Minghui Fang, Jialong Zuo, Qian Yang, Ruiqi Li, Ziang Zhang, Xiaoda Yang, Rongjie Huang, Yidi Jiang, Qian Chen, Siqi Zheng, Wen Wang, and Zhou Zhao. 2024. WavTokenizer: an Efficient Acoustic Discrete Codec Tokenizer for Audio Language Modeling. CoRR, Vol. abs\/2408.16532 (2024)."},{"key":"e_1_3_2_1_28_1","volume-title":"Mega-TTS 2: Zero-Shot Text-to-Speech with Arbitrary Length Speech Prompts. CoRR","author":"Jiang Ziyue","year":"2023","unstructured":"Ziyue Jiang, Jinglin Liu, Yi Ren, Jinzheng He, Chen Zhang, Zhenhui Ye, Pengfei Wei, Chunfeng Wang, Xiang Yin, Zejun Ma, and Zhou Zhao. 2023. Mega-TTS 2: Zero-Shot Text-to-Speech with Arbitrary Length Speech Prompts. CoRR, Vol. abs\/2307.07218 (2023)."},{"key":"e_1_3_2_1_29_1","volume-title":"Forty-first International Conference on Machine Learning, ICML 2024","author":"Ju Zeqian","year":"2024","unstructured":"Zeqian Ju, Yuancheng Wang, Kai Shen, Xu Tan, Detai Xin, Dongchao Yang, Eric Liu, Yichong Leng, Kaitao Song, Siliang Tang, Zhizheng Wu, Tao Qin, Xiangyang Li, Wei Ye, Shikun Zhang, Jiang Bian, Lei He, Jinyu Li, and Sheng Zhao. 2024. NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models. In Forty-first International Conference on Machine Learning, ICML 2024, Vienna, Austria, July 21-27, 2024. OpenReview.net."},{"key":"e_1_3_2_1_30_1","volume-title":"Frechet audio distance: A metric for evaluating music enhancement algorithms. arXiv preprint arXiv:1812.08466","author":"Kilgour Kevin","year":"2018","unstructured":"Kevin Kilgour, Mauricio Zuluaga, Dominik Roblek, and Matthew Sharifi. 2018. Frechet audio distance: A metric for evaluating music enhancement algorithms. arXiv preprint arXiv:1812.08466 (2018)."},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"e_1_3_2_1_32_1","volume-title":"Yifang Yin, and Roger Zimmermann.","author":"Kumar Yaman","year":"2019","unstructured":"Yaman Kumar, Rohit Jain, Khwaja Mohd. Salik, Rajiv Ratn Shah, Yifang Yin, and Roger Zimmermann. 2019. Lipper: Synthesizing Thy Speech Using Multi-View Lipreading. In The Thirty-Third AAAI Conference on Artificial Intelligence, AAAI 2019, The Thirty-First Innovative Applications of Artificial Intelligence Conference, IAAI 2019, The Ninth AAAI Symposium on Educational Advances in Artificial Intelligence, EAAI 2019, Honolulu, Hawaii, USA, January 27 - February 1, 2019. AAAI Press, 2588-2595."},{"key":"e_1_3_2_1_33_1","first-page":"1","volume-title":"Imaginary Voice: Face-Styled Diffusion Model for Text-to-Speech. In IEEE International Conference on Acoustics, Speech and Signal Processing ICASSP 2023","author":"Lee Jiyoung","year":"2023","unstructured":"Jiyoung Lee, Joon Son Chung, and Soo-Whan Chung. 2023. Imaginary Voice: Face-Styled Diffusion Model for Text-to-Speech. In IEEE International Conference on Acoustics, Speech and Signal Processing ICASSP 2023, Rhodes Island, Greece, June 4-10, 2023. IEEE, 1-5."},{"key":"e_1_3_2_1_34_1","first-page":"80107","article-title":"Songcreator: Lyrics-based universal song generation","volume":"37","author":"Lei Shun","year":"2024","unstructured":"Shun Lei, Yixuan Zhou, Boshi Tang, Max WY Lam, Hangyu Liu, Jingcheng Wu, Shiyin Kang, Zhiyong Wu, Helen Meng, et al., 2024. Songcreator: Lyrics-based universal song generation. Advances in Neural Information Processing Systems, Vol. 37 (2024), 80107-80140.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"e_1_3_2_1_35_1","volume-title":"Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023","author":"Li Yinghao Aaron","year":"2023","unstructured":"Yinghao Aaron Li, Cong Han, Vinay S. Raghavan, Gavin Mischler, and Nima Mesgarani. 2023. StyleTTS 2: Towards Human-Level Text-to-Speech through Style Diffusion and Adversarial Training with Large Speech Language Models. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023, Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (Eds.)."},{"key":"e_1_3_2_1_36_1","volume-title":"AudioLDM: Text-to-Audio Generation with Latent Diffusion Models. In International Conference on Machine Learning, ICML 2023","volume":"21474","author":"Liu Haohe","year":"2023","unstructured":"Haohe Liu, Zehua Chen, Yi Yuan, Xinhao Mei, Xubo Liu, Danilo P. Mandic, Wenwu Wang, and Mark D. Plumbley. 2023. AudioLDM: Text-to-Audio Generation with Latent Diffusion Models. In International Conference on Machine Learning, ICML 2023, 23-29 July 2023, Honolulu, Hawaii, USA (Proceedings of Machine Learning Research, Vol. 202), Andreas Krause, Emma Brunskill, Kyunghyun Cho, Barbara Engelhardt, Sivan Sabato, and Jonathan Scarlett (Eds.). PMLR, 21450-21474."},{"key":"e_1_3_2_1_37_1","volume-title":"Diff-Foley: Synchronized Video-to-Audio Synthesis with Latent Diffusion Models. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023","author":"Luo Simian","year":"2023","unstructured":"Simian Luo, Chuanhao Yan, Chenxu Hu, and Hang Zhao. 2023. Diff-Foley: Synchronized Video-to-Audio Synthesis with Latent Diffusion Models. In Advances in Neural Information Processing Systems 36: Annual Conference on Neural Information Processing Systems 2023, NeurIPS 2023, New Orleans, LA, USA, December 10 - 16, 2023, Alice Oh, Tristan Naumann, Amir Globerson, Kate Saenko, Moritz Hardt, and Sergey Levine (Eds.)."},{"key":"e_1_3_2_1_38_1","unstructured":"Lingwei Meng Long Zhou Shujie Liu Sanyuan Chen Bing Han Shujie Hu Yanqing Liu Jinyu Li Sheng Zhao Xixin Wu et al. 2024. Autoregressive speech synthesis without vector quantization. arXiv preprint arXiv:2407.08551 (2024)."},{"key":"e_1_3_2_1_39_1","volume-title":"Proceedings of the 38th International Conference on Machine Learning, ICML 2021","volume":"8763","author":"Radford Alec","year":"2021","unstructured":"Alec Radford, Jong Wook Kim, Chris Hallacy, Aditya Ramesh, Gabriel Goh, Sandhini Agarwal, Girish Sastry, Amanda Askell, Pamela Mishkin, Jack Clark, Gretchen Krueger, and Ilya Sutskever. 2021. Learning Transferable Visual Models From Natural Language Supervision. In Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event (Proceedings of Machine Learning Research, Vol. 139), Marina Meila and Tong Zhang (Eds.). PMLR, 8748-8763."},{"key":"e_1_3_2_1_40_1","volume-title":"International conference on machine learning. PMLR, 28492-28518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust speech recognition via large-scale weak supervision. In International conference on machine learning. PMLR, 28492-28518."},{"key":"e_1_3_2_1_41_1","volume-title":"Utmos: Utokyo-sarulab system for voicemos challenge","author":"Saeki Takaaki","year":"2022","unstructured":"Takaaki Saeki, Detai Xin, Wataru Nakata, Tomoki Koriyama, Shinnosuke Takamichi, and Hiroshi Saruwatari. 2022. Utmos: Utokyo-sarulab system for voicemos challenge 2022. arXiv preprint arXiv:2204.02152 (2022)."},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096023"},{"key":"e_1_3_2_1_43_1","volume-title":"The Twelfth International Conference on Learning Representations, ICLR 2024","author":"Shen Kai","year":"2024","unstructured":"Kai Shen, Zeqian Ju, Xu Tan, Eric Liu, Yichong Leng, Lei He, Tao Qin, Sheng Zhao, and Jiang Bian. 2024. NaturalSpeech 2: Latent Diffusion Models are Natural and Zero-Shot Speech and Singing Synthesizers. In The Twelfth International Conference on Learning Representations, ICLR 2024, Vienna, Austria, May 7-11, 2024. OpenReview.net."},{"key":"e_1_3_2_1_44_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.02589"},{"key":"e_1_3_2_1_45_1","volume-title":"LLaMA: Open and Efficient Foundation Language Models. CoRR","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, Aur\u00e9lien Rodriguez, Armand Joulin, Edouard Grave, and Guillaume Lample. 2023. LLaMA: Open and Efficient Foundation Language Models. CoRR, Vol. abs\/2302.13971 (2023)."},{"key":"e_1_3_2_1_46_1","first-page":"1","volume-title":"Temporally Aligned Audio for Video with Autoregression. In 2025 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2025","author":"Viertola Ilpo","year":"2025","unstructured":"Ilpo Viertola, Vladimir Iashin, and Esa Rahtu. 2025. Temporally Aligned Audio for Video with Autoregression. In 2025 IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2025, Hyderabad, India, April 6-11, 2025. IEEE, 1-5."},{"key":"e_1_3_2_1_47_1","volume-title":"FELLE: Autoregressive Speech Synthesis with Token-Wise Coarse-to-Fine Flow Matching. CoRR","author":"Wang Hui","year":"2025","unstructured":"Hui Wang, Shujie Liu, Lingwei Meng, Jinyu Li, Yifan Yang, Shiwan Zhao, Haiyang Sun, Yanqing Liu, Haoqin Sun, Jiaming Zhou, Yan Lu, and Yong Qin. 2025. FELLE: Autoregressive Speech Synthesis with Token-Wise Coarse-to-Fine Flow Matching. CoRR, Vol. abs\/2502.11128 (2025)."},{"key":"e_1_3_2_1_48_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i14.29475"},{"key":"e_1_3_2_1_49_1","volume-title":"Wei Tsung Lu, and Minz Won","author":"Wang Ju-Chiang","year":"2023","unstructured":"Ju-Chiang Wang, Wei Tsung Lu, and Minz Won. 2023. Mel-Band RoFormer for Music Source Separation. CoRR, Vol. abs\/2310.01809 (2023)."},{"key":"e_1_3_2_1_50_1","volume-title":"Frieren: Efficient Video-to-Audio Generation Network with Rectified Flow Matching. In Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems","author":"Wang Yongqi","year":"2024","unstructured":"Yongqi Wang, Wenxiang Guo, Rongjie Huang, Jiawei Huang, Zehan Wang, Fuming You, Ruiqi Li, and Zhou Zhao. 2024a. Frieren: Efficient Video-to-Audio Generation Network with Rectified Flow Matching. In Advances in Neural Information Processing Systems 38: Annual Conference on Neural Information Processing Systems 2024, NeurIPS 2024, Vancouver, BC, Canada, December 10 - 15, 2024, Amir Globersons, Lester Mackey, Danielle Belgrave, Angela Fan, Ulrich Paquet, Jakub M. Tomczak, and Cheng Zhang (Eds.)."},{"key":"e_1_3_2_1_51_1","first-page":"26866","volume-title":"Sonic VisionLM: Playing Sound with Vision Language Models. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024","author":"Xie Zhifeng","year":"2024","unstructured":"Zhifeng Xie, Shengye Yu, Qile He, and Mengtian Li. 2024. Sonic VisionLM: Playing Sound with Vision Language Models. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, Seattle, WA, USA, June 16-22, 2024. 26866-26875."},{"key":"e_1_3_2_1_52_1","first-page":"7151","volume-title":"Seeing and Hearing: Open-domain Visual-Audio Generation with Diffusion Latent Aligners. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024","author":"Xing Yazhou","year":"2024","unstructured":"Yazhou Xing, Yingqing He, Zeyue Tian, Xintao Wang, and Qifeng Chen. 2024. Seeing and Hearing: Open-domain Visual-Audio Generation with Diffusion Latent Aligners. In IEEE\/CVF Conference on Computer Vision and Pattern Recognition, CVPR 2024, Seattle, WA, USA, June 16-22, 2024. IEEE, 7151-7161."},{"key":"e_1_3_2_1_53_1","unstructured":"An Yang Baosong Yang Binyuan Hui Bo Zheng Bowen Yu Chang Zhou Chengpeng Li Chengyuan Li Dayiheng Liu Fei Huang Guanting Dong Haoran Wei Huan Lin Jialong Tang Jialin Wang Jian Yang Jianhong Tu Jianwei Zhang Jianxin Ma Jianxin Yang Jin Xu Jingren Zhou Jinze Bai Jinzheng He Junyang Lin Kai Dang Keming Lu Keqin Chen Kexin Yang Mei Li Mingfeng Xue Na Ni Pei Zhang Peng Wang Ru Peng Rui Men Ruize Gao Runji Lin Shijie Wang Shuai Bai Sinan Tan Tianhang Zhu Tianhao Li Tianyu Liu Wenbin Ge Xiaodong Deng Xiaohuan Zhou Xingzhang Ren Xinyu Zhang Xipin Wei Xuancheng Ren Xuejing Liu Yang Fan Yang Yao Yichang Zhang Yu Wan Yunfei Chu Yuqiong Liu Zeyu Cui Zhenru Zhang Zhifang Guo and Zhihao Fan. 2024. Qwen2 Technical Report. CoRR Vol. abs\/2407.10671 (2024)."},{"key":"e_1_3_2_1_54_1","volume-title":"UniAudio: An Audio Foundation Model Toward Universal Audio Generation. CoRR","author":"Yang Dongchao","year":"2023","unstructured":"Dongchao Yang, Jinchuan Tian, Xu Tan, Rongjie Huang, Songxiang Liu, Xuankai Chang, Jiatong Shi, Sheng Zhao, Jiang Bian, Xixin Wu, Zhou Zhao, Shinji Watanabe, and Helen Meng. 2023. UniAudio: An Audio Foundation Model Toward Universal Audio Generation. CoRR, Vol. abs\/2310.00704 (2023)."},{"key":"e_1_3_2_1_55_1","volume-title":"Llasa: Scaling Train-Time and Inference-Time Compute for Llama-based Speech Synthesis. arXiv preprint arXiv:2502.04128","author":"Ye Zhen","year":"2025","unstructured":"Zhen Ye, Xinfa Zhu, Chi-Min Chan, Xinsheng Wang, Xu Tan, Jiahe Lei, Yi Peng, Haohe Liu, Yizhu Jin, Zheqi DAI, et al., 2025. Llasa: Scaling Train-Time and Inference-Time Compute for Llama-based Speech Synthesis. arXiv preprint arXiv:2502.04128 (2025)."},{"key":"e_1_3_2_1_56_1","volume-title":"Libritts: A corpus derived from librispeech for text-to-speech. arXiv preprint arXiv:1904.02882","author":"Zen Heiga","year":"2019","unstructured":"Heiga Zen, Viet Dang, Rob Clark, Yu Zhang, Ron J Weiss, Ye Jia, Zhifeng Chen, and Yonghui Wu. 2019. Libritts: A corpus derived from librispeech for text-to-speech. arXiv preprint arXiv:1904.02882 (2019)."},{"key":"e_1_3_2_1_57_1","volume-title":"FoleyCrafter: Bring Silent Videos to Life with Lifelike and Synchronized Sounds. CoRR","author":"Zhang Yiming","year":"2024","unstructured":"Yiming Zhang, Yicheng Gu, Yanhong Zeng, Zhening Xing, Yuancheng Wang, Zhizheng Wu, and Kai Chen. 2024a. FoleyCrafter: Bring Silent Videos to Life with Lifelike and Synchronized Sounds. CoRR, Vol. abs\/2407.01494 (2024)."},{"key":"e_1_3_2_1_58_1","volume-title":"Long-Video Audio Synthesis with Multi-Agent Collaboration. CoRR","author":"Zhang Yehang","year":"2025","unstructured":"Yehang Zhang, Xinli Xu, Xiaojie Xu, Li Liu, and Yingcong Chen. 2025. Long-Video Audio Synthesis with Multi-Agent Collaboration. CoRR, Vol. abs\/2503.10719 (2025)."},{"key":"e_1_3_2_1_59_1","first-page":"7523","volume-title":"Proceedings of the 32nd ACM International Conference on Multimedia, MM 2024","author":"Zhang Zhedong","year":"2024","unstructured":"Zhedong Zhang, Liang Li, Gaoxiang Cong, Haibing Yin, Yuhan Gao, Chenggang Yan, Anton van den Hengel, and Yuankai Qi. 2024b. From Speaker to Dubber: Movie Dubbing with Prosody and Duration Consistency Learning. In Proceedings of the 32nd ACM International Conference on Multimedia, MM 2024, Melbourne, VIC, Australia, 28 October 2024 - 1 November 2024, Jianfei Cai, Mohan S. Kankanhalli, Balakrishnan Prabhakaran, Susanne Boll, Ramanathan Subramanian, Liang Zheng, Vivek K. Singh, Pablo C\u00e9sar, Lexing Xie, and Dong Xu (Eds.). ACM, 7523-7532."},{"key":"e_1_3_2_1_60_1","volume-title":"National Conference on Man-Machine Speech Communication. Springer, 168-182","author":"Zhao Yuan","year":"2024","unstructured":"Yuan Zhao, Zhenqi Jia, Rui Liu, De Hu, Feilong Bao, and Guanglai Gao. 2024. Mcdubber: Multimodal context-aware expressive video dubbing. In National Conference on Man-Machine Speech Communication. Springer, 168-182."},{"key":"e_1_3_2_1_61_1","volume-title":"CosyAudio: Improving Audio Generation with Confidence Scores and Synthetic Captions. CoRR","author":"Zhu Xinfa","year":"2025","unstructured":"Xinfa Zhu, Wenjie Tian, Xinsheng Wang, Lei He, Xi Wang, Sheng Zhao, and Lei Xie. 2025. CosyAudio: Improving Audio Generation with Confidence Scores and Synthetic Captions. CoRR, Vol. abs\/2501.16761 (2025)."},{"key":"e_1_3_2_1_62_1","volume-title":"Autoregressive Speech Synthesis with Next-Distribution Prediction. arXiv preprint arXiv:2412.16846","author":"Zhu Xinfa","year":"2024","unstructured":"Xinfa Zhu, Wenjie Tian, and Lei Xie. 2024. Autoregressive Speech Synthesis with Next-Distribution Prediction. arXiv preprint arXiv:2412.16846 (2024)."}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3755831","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T04:05:19Z","timestamp":1765339519000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3755831"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":62,"alternative-id":["10.1145\/3746027.3755831","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3755831","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}