{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T17:50:06Z","timestamp":1773942606621,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":43,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"DOI":"10.13039\/https:\/\/doi.org\/10.13039\/501100000287","name":"Royal Academy of Engineering","doi-asserted-by":"publisher","award":["RF\\201819\\18\\163"],"award-info":[{"award-number":["RF\\201819\\18\\163"]}],"id":[{"id":"10.13039\/https:\/\/doi.org\/10.13039\/501100000287","id-type":"DOI","asserted-by":"publisher"}]},{"name":"DFG: SFB 1233","award":["276693517"],"award-info":[{"award-number":["276693517"]}]},{"name":"DFG EXC number 2064\/1","award":["390727645"],"award-info":[{"award-number":["390727645"]}]},{"name":"EPSRC","award":["DTA Studentship"],"award-info":[{"award-number":["DTA Studentship"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681690","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:41Z","timestamp":1729925981000},"page":"9535-9543","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["Dissecting Temporal Understanding in Text-to-Audio Retrieval"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9730-3873","authenticated-orcid":false,"given":"Andreea-Maria","family":"Oncescu","sequence":"first","affiliation":[{"name":"Visual Geometry Group, University of Oxford, Oxford, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2478-2102","authenticated-orcid":false,"given":"Jo\u00e3o F.","family":"Henriques","sequence":"additional","affiliation":[{"name":"Visual Geometry Group, University of Oxford, Oxford, United Kingdom"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5807-0576","authenticated-orcid":false,"given":"A. Sophia","family":"Koepke","sequence":"additional","affiliation":[{"name":"T\u00fcbingen AI Center, University of T\u00fcbingen, T\u00fcbingen, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_2_1_1","volume-title":"Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In IEEE International Conference on Computer Vision.","author":"Bain Max","year":"2021","unstructured":"Max Bain, Arsha Nagrani, G\u00fcl Varol, and Andrew Zisserman. 2021. Frozen in Time: A Joint Video and Image Encoder for End-to-End Retrieval. In IEEE International Conference on Computer Vision."},{"key":"e_1_3_2_2_2_1","volume-title":"A CLIP-Hitchhiker's Guide to Long Video Retrieval. arXiv preprint arXiv:2205.08508","author":"Bain Max","year":"2022","unstructured":"Max Bain, Arsha Nagrani, G\u00fcl Varol, and Andrew Zisserman. 2022. A CLIP-Hitchhiker's Guide to Long Video Retrieval. arXiv preprint arXiv:2205.08508 (2022)."},{"key":"e_1_3_2_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096526"},{"key":"e_1_3_2_2_4_1","volume-title":"HTS-AT: A Hierarchical Token-Semantic Audio Transformer for Sound Classification and Detection. International conference on acoustics, speech and signal processing (ICASSP).","author":"Chen Ke","year":"2022","unstructured":"Ke Chen, Xingjian Du, Bilei Zhu, Zejun Ma, Taylor Berg-Kirkpatrick, and Shlomo Dubnov. 2022. HTS-AT: A Hierarchical Token-Semantic Audio Transformer for Sound Classification and Detection. International conference on acoustics, speech and signal processing (ICASSP)."},{"key":"e_1_3_2_2_5_1","volume-title":"VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset. arXiv preprint arXiv:2304.08345","author":"Chen Sihan","year":"2023","unstructured":"Sihan Chen, Xingjian He, Longteng Guo, Xinxin Zhu, Weining Wang, Jinhui Tang, and Jing Liu. 2023. VALOR: Vision-Audio-Language Omni-Perception Pretraining Model and Dataset. arXiv preprint arXiv:2304.08345 (2023)."},{"key":"e_1_3_2_2_6_1","volume-title":"Vast: A vision-audio-subtitle-text omni-modality foundation model and dataset. Advances in Neural Information Processing Systems (NeurIPS)","author":"Chen Sihan","year":"2024","unstructured":"Sihan Chen, Handong Li, Qunbo Wang, Zijia Zhao, Mingzhen Sun, Xinxin Zhu, and Jing Liu. 2024. Vast: A vision-audio-subtitle-text omni-modality foundation model and dataset. Advances in Neural Information Processing Systems (NeurIPS) (2024)."},{"key":"e_1_3_2_2_7_1","volume-title":"International Conference on Machine Learning (ICML).","author":"Chen Ting","year":"2020","unstructured":"Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A simple framework for contrastive learning of visual representations. In International Conference on Machine Learning (ICML)."},{"key":"e_1_3_2_2_8_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448115"},{"key":"e_1_3_2_2_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052990"},{"key":"e_1_3_2_2_10_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095889"},{"key":"e_1_3_2_2_11_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682632"},{"key":"e_1_3_2_2_12_1","volume-title":"Association for Computing Machinery (ACM) International Conference on Multimedia.","author":"Ghias Asif","unstructured":"Asif Ghias, Jonathan Logan, David Chamberlin, and Brian C. Smith. 1995. Query by humming: musical information retrieval in an audio database. In Association for Computing Machinery (ACM) International Conference on Multimedia."},{"key":"e_1_3_2_2_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01720"},{"key":"e_1_3_2_2_14_1","volume-title":"Make-An-Audio: Text-To-Audio Generation with Prompt-Enhanced Diffusion Models. In International Conference on Machine Learning (ICLR).","author":"Huang Rongjie","year":"2023","unstructured":"Rongjie Huang, Jiawei Huang, Dongchao Yang, Yi Ren, Luping Liu, Mingze Li, Zhenhui Ye, Jinglin Liu, Xiang Yin, and Zhou Zhao. 2023. Make-An-Audio: Text-To-Audio Generation with Prompt-Enhanced Diffusion Models. In International Conference on Machine Learning (ICLR)."},{"key":"e_1_3_2_2_15_1","volume-title":"Multi-Modal Dense Video Captioning. In Conference on Computer Vision and Pattern Recognition (CVPR) Workshops.","author":"Iashin Vladimir","year":"2020","unstructured":"Vladimir Iashin and Esa Rahtu. 2020. Multi-Modal Dense Video Captioning. In Conference on Computer Vision and Pattern Recognition (CVPR) Workshops."},{"key":"e_1_3_2_2_16_1","unstructured":"Chris Dongjoo Kim Byeongchang Kim Hyunmin Lee and Gunhee Kim. 2019. AudioCaps: Generating captions for audios in the wild. In North American Chapter of the Association for Computational Linguistics (NAACL)."},{"key":"e_1_3_2_2_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52733.2024.01318"},{"key":"e_1_3_2_2_18_1","volume-title":"Zeynep Akata, and Samuel Albanie.","author":"Koepke A. Sophia","year":"2022","unstructured":"A. Sophia Koepke, Andreea-Maria Oncescu, Jo ao F. Henriques, Zeynep Akata, and Samuel Albanie. 2022. Audio retrieval with natural language queries: A benchmark study. IEEE Transactions on Multimedia (2022)."},{"key":"e_1_3_2_2_19_1","volume-title":"AudioGen: Textually Guided Audio Generation. In International Conference on Learning Representations (ICLR).","author":"Kreuk Felix","year":"2023","unstructured":"Felix Kreuk, Gabriel Synnaeve, Adam Polyak, Uriel Singer, Alexandre D\u00e9fossez, Jade Copet, Devi Parikh, Yaniv Taigman, and Yossi Adi. 2023. AudioGen: Textually Guided Audio Generation. In International Conference on Learning Representations (ICLR)."},{"key":"e_1_3_2_2_20_1","volume-title":"British Machine Vision Conference (BMVC).","author":"Liu Yang","year":"2019","unstructured":"Yang Liu, Samuel Albanie, Arsha Nagrani, and Andrew Zisserman. 2019. Use What You Have: Video retrieval using representations from collaborative experts. In British Machine Vision Conference (BMVC)."},{"key":"e_1_3_2_2_21_1","volume-title":"Uncovering Hidden Challenges in Query-Based Video Moment Retrieval. In The British Machine Vision Conference (BMVC).","author":"Mayu Otani Esa Rahtu","year":"2020","unstructured":"Esa Rahtu Mayu Otani, Yuta Nakahima and Janne Heikkil\u00e4. 2020. Uncovering Hidden Challenges in Query-Based Video Moment Retrieval. In The British Machine Vision Conference (BMVC)."},{"key":"e_1_3_2_2_22_1","volume-title":"Audio Captioning Transformer. In Detection and Classification of Acoustic Scenes and Events 2021 Workshop (DCASE2021)","author":"Mei Xinhao","year":"2021","unstructured":"Xinhao Mei, Xubo Liu, Qiushi Huang, Mark D. Plumbley, and Wenwu Wang. 2021. Audio Captioning Transformer. In Detection and Classification of Acoustic Scenes and Events 2021 Workshop (DCASE2021)."},{"key":"e_1_3_2_2_23_1","doi-asserted-by":"crossref","unstructured":"Xinhao Mei Xubo Liu Jianyuan Sun Mark D. Plumbley and Wenwu Wang. 2022. On Metric Learning for Audio-Text Cross-Modal Retrieval. In INTERSPEECH.","DOI":"10.21437\/Interspeech.2022-11115"},{"key":"e_1_3_2_2_24_1","volume-title":"WavCaps: A ChatGPT-Assisted Weakly-Labelled Audio Captioning Dataset for Audio-Language Multimodal Research. arXiv preprint arXiv:2303.17395","author":"Mei Xinhao","year":"2023","unstructured":"Xinhao Mei, Chutong Meng, Haohe Liu, Qiuqiang Kong, Tom Ko, Chengqi Zhao, Mark D Plumbley, Yuexian Zou, and Wenwu Wang. 2023. WavCaps: A ChatGPT-Assisted Weakly-Labelled Audio Captioning Dataset for Audio-Language Multimodal Research. arXiv preprint arXiv:2303.17395 (2023)."},{"key":"e_1_3_2_2_25_1","volume-title":"International Conference on Acoustics, Speech and Signal Processing (ICASSP).","author":"Oncescu Andreea-Maria","unstructured":"Andreea-Maria Oncescu, Jo ao F. Henriques, Andrew Zisserman, Samuel Albanie, and A. Sophia Koepke. 2024. A Sound Approach: Using Large Language Models to generate audio descriptions for egocentric text-audio retrieval. In International Conference on Acoustics, Speech and Signal Processing (ICASSP)."},{"key":"e_1_3_2_2_26_1","volume-title":"Jo ao F. Henriques, Zeynep Akata, and Samuel Albanie.","author":"Oncescu Andreea-Maria","year":"2021","unstructured":"Andreea-Maria Oncescu, relax A. Sophia Koepke, Jo ao F. Henriques, Zeynep Akata, and Samuel Albanie. 2021. Audio Retrieval with Natural Language Queries. In INTERSPEECH."},{"key":"e_1_3_2_2_27_1","volume-title":"March","author":"AI.","year":"2024","unstructured":"OpenAI. [n.,d.]. GPT. https:\/\/platform.openai.com\/playground\/. Accessed February, March 2024."},{"key":"e_1_3_2_2_28_1","volume-title":"ESC: Dataset for Environmental Sound Classification. In Association for Computing Machinery (ACM) Conference on Multimedia.","author":"Piczak Karol J.","year":"2015","unstructured":"Karol J. Piczak. 2015. ESC: Dataset for Environmental Sound Classification. In Association for Computing Machinery (ACM) Conference on Multimedia."},{"key":"e_1_3_2_2_29_1","volume":"202","author":"Raffel Colin","unstructured":"Colin Raffel, Noam Shazeer, Adam Roberts, Katherine Lee, Sharan Narang, Michael Matena, Yanqi Zhou, Wei Li, and Peter J. Liu. 2020. Exploring the Limits of Transfer Learning with a Unified Text-to-Text Transformer. Journal of Machine Learning Research (2020).","journal-title":"Peter J. Liu."},{"key":"e_1_3_2_2_30_1","volume-title":"International conference on acoustics, speech and signal processing (ICASSP).","author":"Slaney Malcolm","year":"2002","unstructured":"Malcolm Slaney. 2002. Semantic-audio retrieval. In International conference on acoustics, speech and signal processing (ICASSP)."},{"key":"e_1_3_2_2_31_1","volume-title":"ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities. arXiv preprint arXiv:2305.11172","author":"Wang Peng","year":"2023","unstructured":"Peng Wang, Shijie Wang, Junyang Lin, Shuai Bai, Xiaohuan Zhou, Jingren Zhou, Xinggang Wang, and Chang Zhou. 2023. ONE-PEACE: Exploring One General Representation Model Toward Unlimited Modalities. arXiv preprint arXiv:2305.11172 (2023)."},{"key":"e_1_3_2_2_32_1","volume-title":"InternVideo2: Scaling Video Foundation Models for Multimodal Video Understanding. arXiv preprint abs\/2403.15377","author":"Wang Yi","year":"2024","unstructured":"Yi Wang, Kunchang Li, Xinhao Li, Jiashuo Yu, Yinan He, Guo Chen, Baoqi Pei, Rongkun Zheng, Jilan Xu, Zun Wang, Yansong Shi, Tianxiang Jiang, Songze Li, Hongjie Zhang, Yifei Huang, Yu Qiao, Yali Wang, and Limin Wang. 2024. InternVideo2: Scaling Video Foundation Models for Multimodal Video Understanding. arXiv preprint abs\/2403.15377 (2024)."},{"key":"e_1_3_2_2_33_1","doi-asserted-by":"publisher","DOI":"10.1109\/93.556537"},{"key":"e_1_3_2_2_34_1","volume-title":"Audio-Text Models Do Not Yet Leverage Natural Language. International conference on acoustics, speech and signal processing (ICASSP).","author":"Wu Ho-Hsiang","year":"2023","unstructured":"Ho-Hsiang Wu, Oriol Nieto, Juan Pablo Bello, and Justin Salamon. 2023. Audio-Text Models Do Not Yet Leverage Natural Language. International conference on acoustics, speech and signal processing (ICASSP)."},{"key":"e_1_3_2_2_35_1","volume-title":"Large-scale Contrastive Language-Audio Pretraining with Feature Fusion and Keyword-to-Caption Augmentation. In International Conference on Acoustics, Speech and Signal Processing (ICASSP).","author":"Yusong","year":"2023","unstructured":"Yusong Wu*, Ke Chen*, Tianyu Zhang*, Yuchen Hui*, Taylor Berg-Kirkpatrick, and Shlomo Dubnov. 2023. Large-scale Contrastive Language-Audio Pretraining with Feature Fusion and Keyword-to-Caption Augmentation. In International Conference on Acoustics, Speech and Signal Processing (ICASSP)."},{"key":"e_1_3_2_2_36_1","volume-title":"Unsupervised Audio-Caption Aligning Learns Correspondences Between Individual Sound Events and Textual Phrases. In International Conference on Acoustics, Speech and Signal Processing (ICASSP).","author":"Xie Huang","year":"2022","unstructured":"Huang Xie, Okko R\u00e4s\u00e4nen, Konstantinos Drossos, and Tuomas Virtanen. 2022. Unsupervised Audio-Caption Aligning Learns Correspondences Between Individual Sound Events and Textual Phrases. In International Conference on Acoustics, Speech and Signal Processing (ICASSP)."},{"key":"e_1_3_2_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096972"},{"key":"e_1_3_2_2_38_1","volume-title":"Text-to-Audio Grounding: Building Correspondence Between Captions and Sound Events. International conference on acoustics, speech and signal processing (ICASSP).","author":"Xu Xuenan","year":"2021","unstructured":"Xuenan Xu, Heinrich Dinkel, Mengyue Wu, and Kai Yu. 2021. Text-to-Audio Grounding: Building Correspondence Between Captions and Sound Events. International conference on acoustics, speech and signal processing (ICASSP)."},{"key":"e_1_3_2_2_39_1","volume-title":"Towards Weakly Supervised Text-to-Audio Grounding. arXiv preprint arXiv:2401.02584","author":"Xu Xuenan","year":"2024","unstructured":"Xuenan Xu, Ziyang Ma, Mengyue Wu, and Kai Yu. 2024. Towards Weakly Supervised Text-to-Audio Grounding. arXiv preprint arXiv:2401.02584 (2024)."},{"key":"e_1_3_2_2_40_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSPW59220.2023.10192960"},{"key":"e_1_3_2_2_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448441"},{"key":"e_1_3_2_2_42_1","volume-title":"Diffsound: Discrete Diffusion Model for Text-to-Sound Generation","author":"Yang Dongchao","year":"2023","unstructured":"Dongchao Yang, Jianwei Yu, Helin Wang, Wen Wang, Chao Weng, Yuexian Zou, and Dong Yu. 2023. Diffsound: Discrete Diffusion Model for Text-to-Sound Generation. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, Vol. 31 (2023)."},{"key":"e_1_3_2_2_43_1","volume-title":"Flap: Fast Language-Audio Pre-Training. In Automatic Speech Recognition and Understanding Workshop (ASRU).","author":"Yeh Ching-Feng","year":"2023","unstructured":"Ching-Feng Yeh, Po-Yao Huang, Vasu Sharma, Shang-Wen Li, and Gargi Gosh. 2023. Flap: Fast Language-Audio Pre-Training. In Automatic Speech Recognition and Understanding Workshop (ASRU)."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681690","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681690","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:50Z","timestamp":1750295870000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681690"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":43,"alternative-id":["10.1145\/3664647.3681690","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681690","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}