{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T00:28:27Z","timestamp":1765499307156,"version":"3.48.0"},"publisher-location":"New York, NY, USA","reference-count":54,"publisher":"ACM","funder":[{"name":"Lee Kong Chian Fellowship"},{"name":"Faculty of Information Technology, University of Science, Vietnam National University-Ho Chi Minh City"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,11,10]]},"DOI":"10.1145\/3746252.3761269","type":"proceedings-article","created":{"date-parts":[[2025,11,7]],"date-time":"2025-11-07T23:59:18Z","timestamp":1762559958000},"page":"2346-2356","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["DistillCaps: Enhancing Audio-Language Alignment in Captioning via Retrieval-Augmented Knowledge Distillation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-2465-6833","authenticated-orcid":false,"given":"Thinh","family":"Pham","sequence":"first","affiliation":[{"name":"University of Science, Vietnam National University, Ho Chi Minh City, Vietnam"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-7406-1250","authenticated-orcid":false,"given":"Nghiem","family":"Diep","sequence":"additional","affiliation":[{"name":"University of Science, Vietnam National University, Ho Chi Minh City, Vietnam"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9973-3305","authenticated-orcid":false,"given":"Lizi","family":"Liao","sequence":"additional","affiliation":[{"name":"School of Computing and Information Systems, Singapore Management University, Singapore, Singapore"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5249-9702","authenticated-orcid":false,"given":"Binh T.","family":"Nguyen","sequence":"additional","affiliation":[{"name":"AISIA Lab, Ho Chi Minh City, Vietnam and Univerisity of Science, Vietnam National University, Ho Chi Minh City, Vietnam"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2025,11,10]]},"reference":[{"key":"e_1_3_2_2_1_1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46454-1_24"},{"key":"e_1_3_2_2_2_1","volume-title":"Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72","author":"Banerjee Satanjeev","year":"2005","unstructured":"Satanjeev Banerjee and Alon Lavie. 2005. METEOR: An automatic metric for MT evaluation with improved correlation with human judgments. In Proceedings of the acl workshop on intrinsic and extrinsic evaluation measures for machine translation and\/or summarization. 65-72."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSPEC.1969.5213896"},{"key":"e_1_3_2_2_4_1","volume-title":"Audio Captioning via Generative Pair-to-Pair Retrieval with Refined Knowledge Base. arXiv preprint arXiv:2410.10913","author":"Changin Choi","year":"2024","unstructured":"Choi Changin, Lim Sungjun, and Rhee Wonjong. 2024. Audio Captioning via Generative Pair-to-Pair Retrieval with Refined Knowledge Base. arXiv preprint arXiv:2410.10913 (2024)."},{"key":"e_1_3_2_2_5_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746312"},{"key":"e_1_3_2_2_6_1","volume-title":"Proceedings of the Detection and Classification of Acoustic Scenes and Events Workshop (DCASE). 21-25","author":"Chen Kun","year":"2020","unstructured":"Kun Chen, Yusong Wu, Ziyue Wang, Xuan Zhang, Fudong Nian, Shengchen Li, and Xi Shao. 2020. Audio Captioning Based on Transformer and Pre-Trained CNN. In Proceedings of the Detection and Classification of Acoustic Scenes and Events Workshop (DCASE). 21-25."},{"key":"e_1_3_2_2_7_1","volume-title":"Beats: Audio pre-training with acoustic tokenizers. arXiv preprint arXiv:2212.09058","author":"Chen Sanyuan","year":"2022","unstructured":"Sanyuan Chen, Yu Wu, Chengyi Wang, Shujie Liu, Daniel Tompkins, Zhuo Chen, and Furu Wei. 2022b. Beats: Audio pre-training with acoustic tokenizers. arXiv preprint arXiv:2212.09058 (2022)."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.907569"},{"key":"e_1_3_2_2_9_1","first-page":"18090","volume-title":"Advances in Neural Information Processing Systems","volume":"36","author":"Deshmukh Soham","year":"2023","unstructured":"Soham Deshmukh, Benjamin Elizalde, Rita Singh, and Huaming Wang. 2023. Pengi: An Audio Language Model for Audio Tasks. In Advances in Neural Information Processing Systems, Vol. 36. Curran Associates, Inc., 18090-18108. https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/3a2e5889b4bbef997ddb13b55d5acf77-Paper-Conference.pdf"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446348"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-1011"},{"key":"e_1_3_2_2_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2017.8170058"},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052990"},{"key":"e_1_3_2_2_14_1","volume-title":"High fidelity neural audio compression. arXiv preprint arXiv:2210.13438","author":"D\u00e9fossez Alexandre","year":"2022","unstructured":"Alexandre D\u00e9fossez, Jade Copet, Gabriel Synnaeve, and Yossi Adi. 2022. High fidelity neural audio compression. arXiv preprint arXiv:2210.13438 (2022)."},{"key":"e_1_3_2_2_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"e_1_3_2_2_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448030"},{"key":"e_1_3_2_2_18_1","volume-title":"Workshop on Detection and Classification of Acoustic Scenes and Events.","author":"Gontier F\u00e9lix","year":"2021","unstructured":"F\u00e9lix Gontier, Romain Serizel, and Christophe Cerisara. 2021. Automated audio captioning by fine-tuning bart with audioset tags. In Workshop on Detection and Classification of Acoustic Scenes and Events."},{"key":"e_1_3_2_2_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952132"},{"key":"e_1_3_2_2_20_1","volume-title":"Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531","author":"Hinton Geoffrey","year":"2015","unstructured":"Geoffrey Hinton, Oriol Vinyals, and Jeff Dean. 2015. Distilling the knowledge in a neural network. arXiv preprint arXiv:1503.02531 (2015)."},{"key":"e_1_3_2_2_21_1","first-page":"3","article-title":"Lora: Low-rank adaptation of large language models","volume":"1","author":"Hu Edward","year":"2022","unstructured":"Edward Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, and Weizhu Chen. 2022. Lora: Low-rank adaptation of large language models. ICLR, Vol. 1, 2 (2022), 3.","journal-title":"ICLR"},{"key":"e_1_3_2_2_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446672"},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096877"},{"key":"e_1_3_2_2_24_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747676"},{"key":"e_1_3_2_2_25_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"e_1_3_2_2_26_1","volume-title":"Efficient audio captioning transformer with patchout and text guidance. arXiv preprint arXiv:2304.02916","author":"Kouzelis Thodoris","year":"2023","unstructured":"Thodoris Kouzelis, Grigoris Bastas, Athanasios Katsamanis, and Alexandros Potamianos. 2023. Efficient audio captioning transformer with patchout and text guidance. arXiv preprint arXiv:2304.02916 (2023)."},{"key":"e_1_3_2_2_27_1","volume-title":"Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461","author":"Lewis Mike","year":"2019","unstructured":"Mike Lewis, Yinhan Liu, Naman Goyal, Marjan Ghazvininejad, Abdelrahman Mohamed, Omer Levy, Ves Stoyanov, and Luke Zettlemoyer. 2019. Bart: Denoising sequence-to-sequence pre-training for natural language generation, translation, and comprehension. arXiv preprint arXiv:1910.13461 (2019)."},{"key":"e_1_3_2_2_28_1","volume-title":"International conference on machine learning. PMLR","author":"Li Junnan","year":"2023","unstructured":"Junnan Li, Dongxu Li, Silvio Savarese, and Steven Hoi. 2023. Blip-2: Bootstrapping language-image pre-training with frozen image encoders and large language models. In International conference on machine learning. PMLR, 19730-19742."},{"key":"e_1_3_2_2_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49660.2025.10890325"},{"key":"e_1_3_2_2_30_1","volume-title":"Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81.","author":"Lin Chin-Yew","year":"2004","unstructured":"Chin-Yew Lin. 2004. Rouge: A package for automatic evaluation of summaries. In Text summarization branches out. 74-81."},{"key":"e_1_3_2_2_31_1","volume-title":"Enhancing automated audio captioning via large language models with optimized audio encoding. arXiv preprint arXiv:2406.13275","author":"Liu Jizhong","year":"2024","unstructured":"Jizhong Liu, Gang Li, Junbo Zhang, Heinrich Dinkel, Yongqing Wang, Zhiyong Yan, Yujun Wang, and Bin Wang. 2024. Enhancing automated audio captioning via large language models with optimized audio encoding. arXiv preprint arXiv:2406.13275 (2024)."},{"key":"e_1_3_2_2_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.100"},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/TCSI.2019.2933321"},{"key":"e_1_3_2_2_34_1","volume-title":"Audio captioning transformer. arXiv preprint arXiv:2107.09817","author":"Mei Xinhao","year":"2021","unstructured":"Xinhao Mei, Xubo Liu, Qiushi Huang, Mark D. Plumbley, and Wenwu Wang. 2021. Audio captioning transformer. arXiv preprint arXiv:2107.09817 (2021)."},{"key":"e_1_3_2_2_35_1","volume-title":"Automated audio captioning: An overview of recent progress and new challenges. EURASIP journal on audio, speech, and music processing","author":"Mei Xinhao","year":"2022","unstructured":"Xinhao Mei, Xubo Liu, Mark D. Plumbley, and Wenwu Wang. 2022. Automated audio captioning: An overview of recent progress and new challenges. EURASIP journal on audio, speech, and music processing, Vol. 2022, 1 (2022), 26."},{"key":"e_1_3_2_2_36_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSPEC.1970.5213512"},{"key":"e_1_3_2_2_37_1","volume-title":"Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318","author":"Papineni Kishore","year":"2002","unstructured":"Kishore Papineni, Salim Roukos, Todd Ward, and Wei-Jing Zhu. 2002. Bleu: a method for automatic evaluation of machine translation. In Proceedings of the 40th annual meeting of the Association for Computational Linguistics. 311-318."},{"key":"e_1_3_2_2_38_1","volume-title":"Language models are unsupervised multitask learners. OpenAI blog","author":"Radford Alec","year":"2019","unstructured":"Alec Radford, Jeffrey Wu, Rewon Child, David Luan, Dario Amodei, and Ilya Sutskever. 2019. Language models are unsupervised multitask learners. OpenAI blog, Vol. 1, 8 (2019), 9."},{"key":"e_1_3_2_2_39_1","doi-asserted-by":"publisher","DOI":"10.1155\/2007\/65420"},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.33682\/7bay-bj41"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-943"},{"key":"e_1_3_2_2_42_1","volume-title":"Plumbley","author":"Sun Jianyuan","year":"2024","unstructured":"Jianyuan Sun, Wenwu Wang, and Mark D. Plumbley. 2024. PFCA-Net: Pyramid Feature Fusion and Cross Content Attention Network for Automated Audio Captioning. In Interspeech."},{"key":"e_1_3_2_2_43_1","volume-title":"Effects of word-frequency based pre-and post-processings for audio captioning. arXiv preprint arXiv:2009.11436","author":"Takeuchi Daiki","year":"2020","unstructured":"Daiki Takeuchi, Yuma Koizumi, Yasunori Ohishi, Noboru Harada, and Kunio Kashino. 2020. Effects of word-frequency based pre-and post-processings for audio captioning. arXiv preprint arXiv:2009.11436 (2020)."},{"key":"e_1_3_2_2_44_1","volume-title":"International conference on machine learning. PMLR, 6105-6114","author":"Tan Mingxing","year":"2019","unstructured":"Mingxing Tan and Quoc Le. 2019. Efficientnet: Rethinking model scaling for convolutional neural networks. In International conference on machine learning. PMLR, 6105-6114."},{"key":"e_1_3_2_2_45_1","volume-title":"Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971","author":"Touvron Hugo","year":"2023","unstructured":"Hugo Touvron, Thibaut Lavril, Gautier Izacard, Xavier Martinet, Marie-Anne Lachaux, Timoth\u00e9e Lacroix, Baptiste Rozi\u00e8re, Naman Goyal, Eric Hambro, Faisal Azhar, et al., 2023. Llama: Open and efficient foundation language models. arXiv preprint arXiv:2302.13971 (2023)."},{"key":"e_1_3_2_2_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7299087"},{"key":"e_1_3_2_2_47_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682377"},{"key":"e_1_3_2_2_48_1","first-page":"1","article-title":"BEATs-based audio captioning model with INSTRUCTOR embedding supervision and ChatGPT mix-up","author":"Wu Shih-Lun","year":"2023","unstructured":"Shih-Lun Wu, Xuankai Chang, Gordon Wichern, Jee weon Jung, Fran\u00e7ois Germain, Jonathan L. Roux, and Shinji Watanabe. 2023. BEATs-based audio captioning model with INSTRUCTOR embedding supervision and ChatGPT mix-up. In Proceedings of the Detection and Classification of Acoustic Scenes and Events Challenge (DCASE). 1-5.","journal-title":"Proceedings of the Detection and Classification of Acoustic Scenes and Events Challenge (DCASE)."},{"key":"e_1_3_2_2_49_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413982"},{"key":"e_1_3_2_2_50_1","volume-title":"Plumbley","author":"Xu Xuenan","year":"2024","unstructured":"Xuenan Xu, Haohe Liu, Mengyue Wu, Wenwu Wang, and Mark D. Plumbley. 2024a. Efficient Audio Captioning with Encoder-Level Knowledge Distillation. arXiv preprint arXiv:2407.14329 (2024)."},{"key":"e_1_3_2_2_51_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3321968"},{"key":"e_1_3_2_2_52_1","volume-title":"Improving the performance of automated audio captioning via integrating the acoustic and semantic information. arXiv preprint arXiv:2110.06100","author":"Ye Zhongjie","year":"2021","unstructured":"Zhongjie Ye, Helin Wang, Dongchao Yang, and Yuexian Zou. 2021. Improving the performance of automated audio captioning via integrating the acoustic and semantic information. arXiv preprint arXiv:2110.06100 (2021)."},{"key":"e_1_3_2_2_53_1","doi-asserted-by":"publisher","DOI":"10.1109\/IST55454.2022.9827729"},{"key":"e_1_3_2_2_54_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISM.202"}],"event":{"name":"CIKM '25: The 34th ACM International Conference on Information and Knowledge Management","sponsor":["SIGIR ACM Special Interest Group on Information Retrieval","SIGWEB ACM Special Interest Group on Hypertext, Hypermedia, and Web"],"location":"Seoul Republic of Korea","acronym":"CIKM '25"},"container-title":["Proceedings of the 34th ACM International Conference on Information and Knowledge Management"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3746252.3761269","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,12,12]],"date-time":"2025-12-12T00:22:48Z","timestamp":1765498968000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3746252.3761269"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,11,10]]},"references-count":54,"alternative-id":["10.1145\/3746252.3761269","10.1145\/3746252"],"URL":"https:\/\/doi.org\/10.1145\/3746252.3761269","relation":{},"subject":[],"published":{"date-parts":[[2025,11,10]]},"assertion":[{"value":"2025-11-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}