{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,27]],"date-time":"2026-06-27T03:32:08Z","timestamp":1782531128715,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":42,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["No.62272409"],"award-info":[{"award-number":["No.62272409"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758148","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T05:50:47Z","timestamp":1761371447000},"page":"12227-12236","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Singing Timbre Popularity Assessment Based on Multimodal Large Foundation Model"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-5613-3262","authenticated-orcid":false,"given":"Zihao","family":"Wang","sequence":"first","affiliation":[{"name":"Zhejiang University, Hangzhou, China and Carnegie Mellon University, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-0539-6916","authenticated-orcid":false,"given":"Ruibin","family":"Yuan","sequence":"additional","affiliation":[{"name":"Hong Kong University of Science and Technology, Hongkong, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6467-8349","authenticated-orcid":false,"given":"Ziqi","family":"Geng","sequence":"additional","affiliation":[{"name":"University of California, Berkeley, Berkeley, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-6184-9437","authenticated-orcid":false,"given":"Hengjia","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China and Carnegie Mellon University, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3265-7133","authenticated-orcid":false,"given":"Xingwei","family":"Qu","sequence":"additional","affiliation":[{"name":"University of Manchester, Manchester, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-1863-6476","authenticated-orcid":false,"given":"Xinyi","family":"Li","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8401-2370","authenticated-orcid":false,"given":"Songye","family":"Chen","sequence":"additional","affiliation":[{"name":"Mei KTV, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0008-8049-013X","authenticated-orcid":false,"given":"Haoying","family":"Fu","sequence":"additional","affiliation":[{"name":"Mei KTV, Beijing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1823-9856","authenticated-orcid":false,"given":"Roger B.","family":"Dannenberg","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University, Pittsburgh, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0778-2303","authenticated-orcid":false,"given":"Kejun","family":"Zhang","sequence":"additional","affiliation":[{"name":"Zhejiang University, Hangzhou, China and Innovation Center of Yangtze River Delta, Zhejiang University, Hangzhou, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888711"},{"key":"e_1_3_2_2_2_1","volume-title":"Qwen-audio: Advancing universal audio understanding via unified large-scale audio-language models. arXiv preprint arXiv:2311.07919","author":"Chu Yunfei","year":"2023","unstructured":"Yunfei Chu, Jin Xu, Xiaohuan Zhou, Qian Yang, Shiliang Zhang, Zhijie Yan, Chang Zhou, and Jingren Zhou. 2023. Qwen-audio: Advancing universal audio understanding via unified large-scale audio-language models. arXiv preprint arXiv:2311.07919 (2023)."},{"key":"e_1_3_2_2_3_1","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Copet Jade","year":"2024","unstructured":"Jade Copet, Felix Kreuk, Itai Gat, Tal Remez, David Kant, Gabriel Synnaeve, Yossi Adi, and Alexandre D\u00e9fossez. 2024. Simple and controllable music generation. Advances in Neural Information Processing Systems, Vol. 36 (2024)."},{"key":"e_1_3_2_2_4_1","first-page":"1","article-title":"Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Dai Shuqi","year":"2025","unstructured":"Shuqi Dai, Yunyun Wang, Roger B Dannenberg, and Zeyu Jin. 2025. Everyone-Can-Sing: Zero-Shot Singing Voice Synthesis and Conversion with Speech Reference. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 1-5.","journal-title":"IEEE"},{"key":"e_1_3_2_2_5_1","volume-title":"Alec Radford, and Ilya Sutskever.","author":"Dhariwal Prafulla","year":"2020","unstructured":"Prafulla Dhariwal, Heewoo Jun, Christine Payne, Jong Wook Kim, Alec Radford, and Ilya Sutskever. 2020. Jukebox: A generative model for music. arXiv preprint arXiv:2005.00341 (2020)."},{"key":"e_1_3_2_2_6_1","volume-title":"Proceedings of the 21st International Society for Music Information Retrieval Conference (ISMIR). 416-423","author":"Gupta Chitralekha","year":"2020","unstructured":"Chitralekha Gupta, Lin Huang, and Haizhou Li. 2020. Automatic rank-ordering of singing vocals with twin-neural network. In Proceedings of the 21st International Society for Music Information Retrieval Conference (ISMIR). 416-423. https:\/\/archives.ismir.net\/ismir2020\/paper\/000165.pdf"},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1017\/atsip.2018.10"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPAASC53192.2021.9687795"},{"key":"e_1_3_2_2_9_1","first-page":"3","article-title":"Lora: Low-rank adaptation of large language models","volume":"1","author":"Hu Edward J","year":"2022","unstructured":"Edward J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, Weizhu Chen, et al., 2022. Lora: Low-rank adaptation of large language models. ICLR, Vol. 1, 2 (2022), 3.","journal-title":"ICLR"},{"key":"e_1_3_2_2_10_1","volume-title":"2020 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC). 492-499","author":"Huang Lin","year":"2020","unstructured":"Lin Huang, Chitralekha Gupta, and Haizhou Li. 2020. Spectral features and pitch histogram for automatic singing quality evaluation with CRNN. In 2020 Asia-Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC). 492-499. http:\/\/www.apsipa.org\/proceedings\/2020\/pdfs\/0000492.pdf"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME55011.2023.00111"},{"key":"e_1_3_2_2_12_1","volume-title":"Proceedings of the 25th International Society for Music Information Retrieval Conference (ISMIR)","author":"Ju Yaolong","year":"2024","unstructured":"Yaolong Ju, Jing Yang, Chun Yat Wu, Betty Corti nas Lorenzo, Jiajun Deng, Fan Fan, and Simon Lui. 2024. End-to-end automatic singing skill evaluation using cross-attention and data augmentation for solo singing and singing with accompaniment. In Proceedings of the 25th International Society for Music Information Retrieval Conference (ISMIR). San Francisco, United States. Forthcoming, URL: https:\/\/www.researchgate.net\/publication\/389749806."},{"key":"e_1_3_2_2_13_1","first-page":"295","volume-title":"Proceedings of the 20th International Society for Music Information Retrieval Conference (ISMIR)","author":"Lee Kyungyun","year":"2019","unstructured":"Kyungyun Lee and Juhan Nam. 2019. Learning a Joint Embedding Space of Monophonic and Mixed Music Signals for Singing Voice. In Proceedings of the 20th International Society for Music Information Retrieval Conference (ISMIR). Delft, The Netherlands, 295-302."},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.23919\/APSIPAASC55919.2022.9980293"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPAASC53192.2021.9687793"},{"key":"e_1_3_2_2_16_1","unstructured":"Zheng Lian Haoyu Chen Lan Chen Haiyang Sun Licai Sun Yong Ren Zebang Cheng Bin Liu Rui Liu Xiaojiang Peng Jiangyan Yi and Jianhua Tao. 2025. AffectGPT: A New Dataset Model and Benchmark for Emotion Understanding with Multimodal Large Language Models. arXiv:2501.16566 [cs.HC] https:\/\/arxiv.org\/abs\/2501.16566"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICIP49359.2023.10222365"},{"key":"e_1_3_2_2_18_1","volume-title":"The Oxford Handbook of Music Psychology","author":"McAdams Stephen","unstructured":"Stephen McAdams. 2009. The perception of musical timbre. In The Oxford Handbook of Music Psychology, Susan Hallam, Ian Cross, and Michael Thaut (Eds.). Oxford University Press, 72-80."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2006-474"},{"key":"e_1_3_2_2_20_1","first-page":"1507","volume-title":"Proceedings of the 9th International Conference on Music Perception and Cognition (ICMPC9), Mario Baroni, Anna Rita Addessi, Roberto Caterina, and Marco Costa (Eds.)","author":"Nakano Tomoyasu","year":"2006","unstructured":"Tomoyasu Nakano, Masataka Goto, and Yuzuru Hiraga. 2006b. Subjective evaluation of common singing skills using the rank ordering method. In Proceedings of the 9th International Conference on Music Perception and Cognition (ICMPC9), Mario Baroni, Anna Rita Addessi, Roberto Caterina, and Marco Costa (Eds.). Bologna, Italy, 1507-1512."},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.1016\/s0892-1997(96)80003-8"},{"key":"e_1_3_2_2_22_1","volume-title":"Proceedings of the 40th International Conference on Machine Learning, ICML 2023","volume":"28518","author":"Radford Alec","year":"2023","unstructured":"Alec Radford, Jong Wook Kim, Tao Xu, Greg Brockman, Christine McLeavey, and Ilya Sutskever. 2023. Robust Speech Recognition via Large-Scale Weak Supervision. In Proceedings of the 40th International Conference on Machine Learning, ICML 2023, 23-29 July 2023, Honolulu, Hawaii, USA (Proceedings of Machine Learning Research, Vol. 202). PMLR, 28492-28518."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096309"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.1914609"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5946974"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2174200"},{"key":"e_1_3_2_2_28_1","volume-title":"Attention is all you need. Advances in neural information processing systems","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems, Vol. 30 (2017)."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096626"},{"key":"e_1_3_2_2_30_1","unstructured":"Zihao Wang Shuyu Li Tao Zhang Qi Wang Pengfei Yu Jinyang Luo Yan Liu Ming Xi and Kejun Zhang. 2024a. MuChin: A Chinese Colloquial Description Benchmark for Evaluating Language Models in the Field of Music. arXiv:2402.09871 [cs.SD] https:\/\/arxiv.org\/abs\/2402.09871"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2024\/860"},{"key":"e_1_3_2_2_32_1","volume-title":"Alignment with Colloquial Expression in Description-to-Song Generation. arXiv preprint arXiv:2407.03188","author":"Wang Zihao","year":"2024","unstructured":"Zihao Wang, Haoxuan Liu, Jiaxing Yu, Tao Zhang, Yan Liu, and Kejun Zhang. 2024c. MuDiT & MuSiT: Alignment with Colloquial Expression in Description-to-Song Generation. arXiv preprint arXiv:2407.03188 (2024)."},{"key":"e_1_3_2_2_33_1","unstructured":"Zihao Wang Le Ma Yongsheng Feng Xin Pan Yuhang Jin and Kejun Zhang. 2024d. SaMoye: Zero-shot Singing Voice Conversion Model Based on Feature Disentanglement and Enhancement. arXiv:2407.07728 [cs.SD] https:\/\/arxiv.org\/abs\/2407.07728"},{"key":"e_1_3_2_2_34_1","unstructured":"Zihao Wang Le Ma Chen Zhang Bo Han Yunfei Xu Yikai Wang Xinyi Chen HaoRong Hong Wenbo Liu Xinda Wu and Kejun Zhang. 2024 e. REMAST: Real-time Emotion-based Music Arrangement with Soft Transition. arXiv:2305.08029 [cs.SD] https:\/\/arxiv.org\/abs\/2305.08029"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TAFFC.2024.3486224"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2025.findings-acl.360"},{"key":"e_1_3_2_2_37_1","volume-title":"SongDriver: Real-time Music Accompaniment Generation without Logical Latency nor Exposure Bias. arXiv preprint arXiv:2209.06054","author":"Wang Zihao","year":"2022","unstructured":"Zihao Wang, Kejun Zhang, Yuxing Wang, Chen Zhang, Qihao Liang, Pengfei Yu, Yongsheng Feng, Wenbo Liu, Yikai Wang, Yuntao Bao, and Yiheng Yang. 2022a. SongDriver: Real-time Music Accompaniment Generation without Logical Latency nor Exposure Bias. arXiv preprint arXiv:2209.06054 (2022)."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3548368"},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1016\/s0892-1997(97)80039-2"},{"key":"e_1_3_2_2_40_1","volume-title":"MelodyGLM: multi-task pre-training for symbolic melody generation. arXiv preprint arXiv:2309.10738","author":"Wu Xinda","year":"2023","unstructured":"Xinda Wu, Zhijie Huang, Kejun Zhang, Jiaxing Yu, Xu Tan, Tieyao Zhang, Zihao Wang, and Lingyun Sun. 2023. MelodyGLM: multi-task pre-training for symbolic melody generation. arXiv preprint arXiv:2309.10738 (2023)."},{"key":"e_1_3_2_2_41_1","unstructured":"Ruibin Yuan Hanfeng Lin Shawn Guo Ge Zhang Jiahao Pan Yongyi Zang Haohe Liu Xingjian Du Xeron Du Zhen Ye Tianyu Zheng Yinghao Ma Minghao Liu Lijun Yu Zeyue Tian Ziya Zhou Liumeng Xue Xingwei Qu Yizhi Li Tianhao Shen Ziyang Ma Shangda Wu Jun Zhan Chunhui Wang Yatian Wang Xiaohuan Zhou Xiaowei Chi Xinyue Zhang Zhenzhu Yang Yiming Liang Xiangzhou Wang Shansong Liu Lingrui Mei Peng Li Yong Chen Chenghua Lin Xie Chen Gus Xia Zhaoxiang Zhang Chao Zhang Wenhu Chen Xinyu Zhou Xipeng Qiu Roger Dannenberg Jiaheng Liu Jian Yang Stephen Huang Wei Xue Xu Tan and Yike Guo. 2025. YuE: Open Music Foundation Models for Full-Song Generation. https:\/\/github.com\/multimodal-art-projection\/YuE. GitHub repository."},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682665"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758148","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,9]],"date-time":"2025-12-09T19:19:27Z","timestamp":1765307967000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758148"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":42,"alternative-id":["10.1145\/3746027.3758148","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758148","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}