{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,16]],"date-time":"2026-06-16T05:07:05Z","timestamp":1781586425160,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":69,"publisher":"ACM","funder":[{"DOI":"10.13039\/501100018537","name":"National Science and Technology Major Project","doi-asserted-by":"publisher","award":["2023ZD0121101"],"award-info":[{"award-number":["2023ZD0121101"]}],"id":[{"id":"10.13039\/501100018537","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100007085","name":"National University of Defense Technology","doi-asserted-by":"publisher","award":["ZZCX-ZZGC-01-04"],"award-info":[{"award-number":["ZZCX-ZZGC-01-04"]}],"id":[{"id":"10.13039\/501100007085","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Major Fundamental Research Project of Hunan Province","award":["2025JC0005"],"award-info":[{"award-number":["2025JC0005"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,10,27]]},"DOI":"10.1145\/3746027.3758260","type":"proceedings-article","created":{"date-parts":[[2025,10,25]],"date-time":"2025-10-25T07:37:21Z","timestamp":1761377841000},"page":"13089-13096","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["AudioSet-R: A Refined AudioSet with Multi-Stage LLM Label Reannotation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-3717-2223","authenticated-orcid":false,"given":"Yulin","family":"Sun","sequence":"first","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4141-5950","authenticated-orcid":false,"given":"Qisheng","family":"Xu","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-8794-7376","authenticated-orcid":false,"given":"Yi","family":"Su","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-8663-0382","authenticated-orcid":false,"given":"Qian","family":"Zhu","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1256-8934","authenticated-orcid":false,"given":"Yong","family":"Dou","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9066-1475","authenticated-orcid":false,"given":"Xinwang","family":"Liu","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5997-5169","authenticated-orcid":false,"given":"Kele","family":"Xu","sequence":"additional","affiliation":[{"name":"College of Computer Science and Technology, National University of Defense Technology, Changsha, Hunan, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2025,10,27]]},"reference":[{"key":"e_1_3_2_2_1_1","first-page":"11","volume-title":"AudioSetCaps: Enriched Audio Captioning Dataset Generation Using Large Audio Language Models. In Audio Imagination: NeurIPS 2024 Workshop AI-Driven Speech, Music, and Sound Generation","volume":"1","author":"Bai Jisheng","year":"2024","unstructured":"Jisheng Bai, Haohe Liu, Mou Wang, Dongyuan Shi, Wenwu Wang, Mark D Plumbley, Woon-Seng Gan, and Jianfeng Chen. 2024. AudioSetCaps: Enriched Audio Captioning Dataset Generation Using Large Audio Language Models. In Audio Imagination: NeurIPS 2024 Workshop AI-Driven Speech, Music, and Sound Generation, Vol. 1. IEEE, Vancouver, Canada, 11-24. https:\/\/openreview.net\/forum?id=uez4PMZwzP"},{"key":"e_1_3_2_2_2_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW63382.2024.00127"},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICTBIG64922.2024.10911672"},{"key":"e_1_3_2_2_4_1","first-page":"1","article-title":"A Systematic Review of Heart Sound Detection Algorithms: Experimental Results and Insights","volume":"74","author":"Bi Xiuli","year":"2025","unstructured":"Xiuli Bi, Shizhan Tang, Bin Xiao, Weisheng Li, Xinbo Gao, and Pietro Li\u00f2. 2025. A Systematic Review of Heart Sound Detection Algorithms: Experimental Results and Insights. IEEE Transactions on Instrumentation and Measurement, Vol. 74 (2025), 1-16.","journal-title":"IEEE Transactions on Instrumentation and Measurement"},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2024-714"},{"key":"e_1_3_2_2_6_1","volume-title":"Beats: Audio pre-training with acoustic tokenizers. arXiv preprint arXiv:2212.09058","author":"Chen Sanyuan","year":"2022","unstructured":"Sanyuan Chen, Yu Wu, Chengyi Wang, Shujie Liu, Daniel Tompkins, Zhuo Chen, and Furu Wei. 2022. Beats: Audio pre-training with acoustic tokenizers. arXiv preprint arXiv:2212.09058, Vol. 1 (2022), 1-5."},{"key":"e_1_3_2_2_7_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11263-025-02358-x"},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW67362.2025.00318"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2598304"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1145\/3729471"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPRW67362.2025.00223"},{"key":"e_1_3_2_2_12_1","volume-title":"Qwen-audio: Advancing universal audio understanding via unified large-scale audio-language models. arXiv preprint arXiv:2311.07919","author":"Chu Yunfei","year":"2023","unstructured":"Yunfei Chu, Jin Xu, Xiaohuan Zhou, Qian Yang, Shiliang Zhang, Zhijie Yan, Chang Zhou, and Jingren Zhou. 2023. Qwen-audio: Advancing universal audio understanding via unified large-scale audio-language models. arXiv preprint arXiv:2311.07919, Vol. 1 (2023), 1."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746431"},{"key":"e_1_3_2_2_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-021-11610-8"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_2_17_1","volume-title":"Utkarsh Tyagi, S Sakshi, Oriol Nieto, Ramani Duraiswami, and Dinesh Manocha.","author":"Ghosh Sreyan","year":"2024","unstructured":"Sreyan Ghosh, Sonal Kumar, Ashish Seth, Chandra Kiran Reddy Evuru, Utkarsh Tyagi, S Sakshi, Oriol Nieto, Ramani Duraiswami, and Dinesh Manocha. 2024. GAMA: A Large Audio-Language Model with Advanced Audio Understanding and Complex Reasoning Abilities. In EMNLP. Association for Computational Linguistics, Miami, Florida, USA, 6288-6313."},{"key":"e_1_3_2_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3350893"},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-698"},{"key":"e_1_3_2_2_20_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21315"},{"key":"e_1_3_2_2_21_1","doi-asserted-by":"publisher","DOI":"10.23919\/SPA61993.2024.10715611"},{"key":"e_1_3_2_2_22_1","unstructured":"Daya Guo Dejian Yang Haowei Zhang Junxiao Song Ruoyu Zhang Runxin Xu Qihao Zhu Shirong Ma Peiyi Wang Xiao Bi et al. 2025. Deepseek-r1: Incentivizing reasoning capability in llms via reinforcement learning. arXiv preprint arXiv:2501.12948 Vol. 1 (2025) 1."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.47760\/ijcsmc.2025.v14i02.001"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"e_1_3_2_2_25_1","first-page":"28708","volume-title":"Oh (Eds.)","volume":"35","author":"Huang Po-Yao","year":"2022","unstructured":"Po-Yao Huang, Hu Xu, Juncheng Li, Alexei Baevski, Michael Auli, Wojciech Galuba, Florian Metze, and Christoph Feichtenhofer. 2022. Masked Autoencoders that Listen. In Advances in Neural Information Processing Systems, S. Koyejo, S. Mohamed, A. Agarwal, D. Belgrave, K. Cho, and A. Oh (Eds.), Vol. 35. Curran Associates, Inc., New Orleans, USA, 28708-28720. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2022\/file\/b89d5e209990b19e33b418e14f323998-Paper-Conference.pdf"},{"key":"e_1_3_2_2_26_1","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO63174.2024.10715018"},{"key":"e_1_3_2_2_27_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICTBIG64922.2024.10911772"},{"key":"e_1_3_2_2_28_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICEES61253.2024.10776884"},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1111\/brv.13155"},{"key":"e_1_3_2_2_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"e_1_3_2_2_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3430813"},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1121\/10.0034981"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10887618"},{"key":"e_1_3_2_2_34_1","doi-asserted-by":"publisher","DOI":"10.1145\/3627704"},{"key":"e_1_3_2_2_35_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3533375"},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3533375"},{"key":"e_1_3_2_2_37_1","unstructured":"Haoyu Lu Wen Liu Bo Zhang Bingxuan Wang Kai Dong Bo Liu Jingxiang Sun Tongzheng Ren Zhuoshu Li Hao Yang et al. 2024. Deepseek-vl: towards real-world vision-language understanding. arXiv preprint arXiv:2403.05525 Vol. 1 (2024) 949-962."},{"key":"e_1_3_2_2_38_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447952"},{"key":"e_1_3_2_2_39_1","first-page":"1","volume-title":"Domain-Incremental Learning for Audio Classification. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, IEEE","author":"Mulimani Manjunath","year":"2025","unstructured":"Manjunath Mulimani and Annamaria Mesaros. 2025. Domain-Incremental Learning for Audio Classification. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, IEEE, Hyderabad, India, 1-5."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2019.2908700"},{"key":"e_1_3_2_2_41_1","first-page":"1","article-title":"Masked Latent Prediction and Classification for Self-Supervised Audio Representation Learning. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, IEEE, Hyderabad","author":"Quelennec Aurian","year":"2025","unstructured":"Aurian Quelennec, Pierre Chouteau, Geoffroy Peeters, and Slim Essid. 2025. Masked Latent Prediction and Classification for Self-Supervised Audio Representation Learning. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, IEEE, Hyderabad, India, 1-5.","journal-title":"India"},{"key":"e_1_3_2_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10888942"},{"key":"e_1_3_2_2_43_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447229"},{"key":"e_1_3_2_2_44_1","first-page":"1","volume-title":"Self-Supervised Learning-Based General Fine-tuning Framework For Audio Classification and Event Detection. In 2024 IEEE International Conference on Multimedia and Expo (ICME). IEEE, IEEE","author":"Sun Yanjie","year":"2024","unstructured":"Yanjie Sun, Kele Xu, Yong Dou, and Tian Gao. 2024a. Self-Supervised Learning-Based General Fine-tuning Framework For Audio Classification and Event Detection. In 2024 IEEE International Conference on Multimedia and Expo (ICME). IEEE, IEEE, Niagra Falls, Canada, 1-6."},{"key":"e_1_3_2_2_45_1","doi-asserted-by":"publisher","DOI":"10.1145\/3581783.3612849"},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3402049"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1007\/s10115-024-02283-4"},{"key":"e_1_3_2_2_48_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICERCS57948.2023.10434081"},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10445941"},{"key":"e_1_3_2_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/3664647.3688999"},{"key":"e_1_3_2_2_51_1","volume-title":"ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Tuncay Ludovic","unstructured":"Ludovic Tuncay, Etienne Labb\u00e9, and Thomas Pellegrini. 2025. Hierarchical Label Propagation: A Model-Size-Dependent Performance Booster for AudioSet Tagging. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, IEEE, Hyderabad,India, 1-5."},{"key":"e_1_3_2_2_52_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890590"},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.5555\/3600270.3602070"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491101.3519729"},{"key":"e_1_3_2_2_55_1","doi-asserted-by":"publisher","DOI":"10.1145\/3491102.3517582"},{"key":"e_1_3_2_2_56_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095969"},{"key":"e_1_3_2_2_57_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.4988946"},{"key":"e_1_3_2_2_58_1","doi-asserted-by":"publisher","DOI":"10.1121\/10.0019937"},{"key":"e_1_3_2_2_59_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.5111059"},{"key":"e_1_3_2_2_60_1","doi-asserted-by":"publisher","DOI":"10.1109\/JSTARS.2023.3325921"},{"key":"e_1_3_2_2_61_1","doi-asserted-by":"publisher","DOI":"10.1121\/10.0016332"},{"key":"e_1_3_2_2_62_1","doi-asserted-by":"publisher","DOI":"10.1145\/3503161.3551579"},{"key":"e_1_3_2_2_63_1","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3318015"},{"key":"e_1_3_2_2_64_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.patcog.2025.111432"},{"key":"e_1_3_2_2_65_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i24.34808"},{"key":"e_1_3_2_2_66_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v39i10.33185"},{"key":"e_1_3_2_2_67_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3008832"},{"key":"e_1_3_2_2_68_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3485170"},{"key":"e_1_3_2_2_69_1","first-page":"1","article-title":"SSAST-Adapter: A Parameter-efficient Incremental Learning Algorithm for Underwater Acoustic Target Recognition. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, IEEE, Hyderabad","author":"Zhu Qian","year":"2025","unstructured":"Qian Zhu, Qisheng Xu, Boqing Zhu, Zijian Gao, Lingbin Zeng, and Kele Xu. 2025. SSAST-Adapter: A Parameter-efficient Incremental Learning Algorithm for Underwater Acoustic Target Recognition. In ICASSP 2025-2025 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, IEEE, Hyderabad, India, 1-5.","journal-title":"India"}],"event":{"name":"MM '25: The 33rd ACM International Conference on Multimedia","location":"Dublin Ireland","acronym":"MM '25","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 33rd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746027.3758260","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,10]],"date-time":"2025-12-10T05:00:11Z","timestamp":1765342811000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746027.3758260"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,27]]},"references-count":69,"alternative-id":["10.1145\/3746027.3758260","10.1145\/3746027"],"URL":"https:\/\/doi.org\/10.1145\/3746027.3758260","relation":{},"subject":[],"published":{"date-parts":[[2025,10,27]]},"assertion":[{"value":"2025-10-27","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}