{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T17:49:53Z","timestamp":1773942593991,"version":"3.50.1"},"publisher-location":"New York, NY, USA","reference-count":32,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,3,17]],"date-time":"2023-03-17T00:00:00Z","timestamp":1679011200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,3,17]]},"DOI":"10.1145\/3590003.3590056","type":"proceedings-article","created":{"date-parts":[[2023,5,29]],"date-time":"2023-05-29T18:22:56Z","timestamp":1685384576000},"page":"298-304","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":1,"title":["Cross-Modal Audio-Text Retrieval via Sequential Feature Augmentation"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0007-5017-5430","authenticated-orcid":false,"given":"Fuhu","family":"Song","sequence":"first","affiliation":[{"name":"Jilin University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-8658-9447","authenticated-orcid":false,"given":"Jifeng","family":"Hu","sequence":"additional","affiliation":[{"name":"Jilin University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0009-5171-9553","authenticated-orcid":false,"given":"Che","family":"Wang","sequence":"additional","affiliation":[{"name":"Jilin University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-8776-0186","authenticated-orcid":false,"given":"Jiao","family":"Huang","sequence":"additional","affiliation":[{"name":"Jilin University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0001-4840-2709","authenticated-orcid":false,"given":"Haowen","family":"Zhang","sequence":"additional","affiliation":[{"name":"Jilin University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-6681-608X","authenticated-orcid":false,"given":"Yi","family":"Wang","sequence":"additional","affiliation":[{"name":"Jilin University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,5,29]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/72.279181"},{"key":"e_1_3_2_1_2_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2022\/759"},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1145\/1460096.1460115"},{"key":"e_1_3_2_1_4_1","volume-title":"International conference on machine learning, Vol.\u00a0119","author":"Chen Ting","year":"2020","unstructured":"Ting Chen , Simon Kornblith , Mohammad Norouzi , and Geoffrey Hinton . 2020 . A Simple Framework for Contrastive Learning of Visual Representations . In International conference on machine learning, Vol.\u00a0119 . PMLR, PMLR, 1597\u20131607. Ting Chen, Simon Kornblith, Mohammad Norouzi, and Geoffrey Hinton. 2020. A Simple Framework for Contrastive Learning of Visual Representations. In International conference on machine learning, Vol.\u00a0119. PMLR, PMLR, 1597\u20131607."},{"key":"e_1_3_2_1_5_1","volume-title":"On the properties of neural machine translation: Encoder-decoder approaches. arXiv preprint arXiv:1409.1259","author":"Cho Kyunghyun","year":"2014","unstructured":"Kyunghyun Cho , Bart Van\u00a0Merri\u00ebnboer , Dzmitry Bahdanau , and Yoshua Bengio . 2014. On the properties of neural machine translation: Encoder-decoder approaches. arXiv preprint arXiv:1409.1259 ( 2014 ). Kyunghyun Cho, Bart Van\u00a0Merri\u00ebnboer, Dzmitry Bahdanau, and Yoshua Bengio. 2014. On the properties of neural machine translation: Encoder-decoder approaches. arXiv preprint arXiv:1409.1259 (2014)."},{"key":"e_1_3_2_1_6_1","volume-title":"Empirical evaluation of gated recurrent neural networks on sequence modeling. CoRR abs\/1412.3555","author":"Chung Junyoung","year":"2014","unstructured":"Junyoung Chung , Caglar Gulcehre , Kyunghyun Cho , and Yoshua Bengio . 2014. Empirical evaluation of gated recurrent neural networks on sequence modeling. CoRR abs\/1412.3555 ( 2014 ). Junyoung Chung, Caglar Gulcehre, Kyunghyun Cho, and Yoshua Bengio. 2014. Empirical evaluation of gated recurrent neural networks on sequence modeling. CoRR abs\/1412.3555 (2014)."},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"crossref","unstructured":"Soham Deshmukh Benjamin Elizalde and Huaming Wang. 2022. Audio Retrieval with WavText5K and CLAP Training. arxiv:2209.14275  Soham Deshmukh Benjamin Elizalde and Huaming Wang. 2022. Audio Retrieval with WavText5K and CLAP Training. arxiv:2209.14275","DOI":"10.21437\/Interspeech.2023-1136"},{"key":"e_1_3_2_1_8_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Devlin Jacob","year":"2019","unstructured":"Jacob Devlin , Ming-Wei Chang , Kenton Lee , and Kristina Toutanova . 2019 . BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding . In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies , Volume 1 (Long and Short Papers). Association for Computational Linguistics, 4171\u20134186. Jacob Devlin, Ming-Wei Chang, Kenton Lee, and Kristina Toutanova. 2019. BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers). Association for Computational Linguistics, 4171\u20134186."},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052990"},{"key":"e_1_3_2_1_10_1","volume-title":"Cross Modal Audio Search and Retrieval with Joint Embeddings Based on Text and Audio. In ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 4095\u20134099","author":"Elizalde Benjamin","year":"2019","unstructured":"Benjamin Elizalde , Shuayb Zarar , and Bhiksha Raj . 2019 . Cross Modal Audio Search and Retrieval with Joint Embeddings Based on Text and Audio. In ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 4095\u20134099 . Benjamin Elizalde, Shuayb Zarar, and Bhiksha Raj. 2019. Cross Modal Audio Search and Retrieval with Joint Embeddings Based on Text and Audio. In ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 4095\u20134099."},{"key":"e_1_3_2_1_11_1","volume-title":"Finding structure in time. Cognitive science 14, 2","author":"Elman L","year":"1990","unstructured":"Jeffrey\u00a0 L Elman . 1990. Finding structure in time. Cognitive science 14, 2 ( 1990 ), 179\u2013211. Jeffrey\u00a0L Elman. 1990. Finding structure in time. Cognitive science 14, 2 (1990), 179\u2013211."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1145\/2502081.2502245"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_1_14_1","volume-title":"Long short-term memory. Neural computation 9, 8","author":"Hochreiter Sepp","year":"1997","unstructured":"Sepp Hochreiter and J\u00fcrgen Schmidhuber . 1997. Long short-term memory. Neural computation 9, 8 ( 1997 ), 1735\u20131780. Sepp Hochreiter and J\u00fcrgen Schmidhuber. 1997. Long short-term memory. Neural computation 9, 8 (1997), 1735\u20131780."},{"key":"e_1_3_2_1_15_1","volume-title":"Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval. In European Conference on Computer Vision. Springer, 444\u2013461","author":"Hu Fan","year":"2022","unstructured":"Fan Hu , Aozhu Chen , Ziyue Wang , Fangming Zhou , Jianfeng Dong , and Xirong Li . 2022 . Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval. In European Conference on Computer Vision. Springer, 444\u2013461 . Fan Hu, Aozhu Chen, Ziyue Wang, Fangming Zhou, Jianfeng Dong, and Xirong Li. 2022. Lightweight Attentional Feature Fusion: A New Baseline for Text-to-Video Retrieval. In European Conference on Computer Vision. Springer, 444\u2013461."},{"key":"e_1_3_2_1_16_1","volume-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","volume":"1","author":"Kim Chris\u00a0Dongjoo","year":"2019","unstructured":"Chris\u00a0Dongjoo Kim , Byeongchang Kim , Hyunmin Lee , and Gunhee Kim . 2019 . Audiocaps: Generating captions for audios in the wild . In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies , Volume 1 (Long and Short Papers). 119\u2013132. Chris\u00a0Dongjoo Kim, Byeongchang Kim, Hyunmin Lee, and Gunhee Kim. 2019. Audiocaps: Generating captions for audios in the wild. In Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers). 119\u2013132."},{"key":"e_1_3_2_1_17_1","volume-title":"Kingma and Jimmy Ba","author":"P.","year":"2015","unstructured":"Diederik\u00a0 P. Kingma and Jimmy Ba . 2015 . Adam : A Method for Stochastic Optimization. In 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings, Yoshua Bengio and Yann LeCun (Eds .). Diederik\u00a0P. Kingma and Jimmy Ba. 2015. Adam: A Method for Stochastic Optimization. In 3rd International Conference on Learning Representations, ICLR 2015, San Diego, CA, USA, May 7-9, 2015, Conference Track Proceedings, Yoshua Bengio and Yann LeCun (Eds.)."},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2022.3149712"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3030497"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.100"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2017.442"},{"key":"e_1_3_2_1_22_1","volume-title":"Audio-Text Retrieval in Context. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 4793\u20134797","author":"Lou Siyu","year":"2022","unstructured":"Siyu Lou , Xuenan Xu , Mengyue Wu , and Kai Yu . 2022 . Audio-Text Retrieval in Context. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 4793\u20134797 . Siyu Lou, Xuenan Xu, Mengyue Wu, and Kai Yu. 2022. Audio-Text Retrieval in Context. In ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 4793\u20134797."},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-11115"},{"key":"e_1_3_2_1_24_1","doi-asserted-by":"crossref","unstructured":"Andreea-Maria Oncescu A.\u00a0Sophia Koepke Jo\u00e3o\u00a0F. Henriques Zeynep Akata and Samuel Albanie. 2021. Audio Retrieval with Natural Language Queries. In INTERSPEECH. 2411\u20132415.  Andreea-Maria Oncescu A.\u00a0Sophia Koepke Jo\u00e3o\u00a0F. Henriques Zeynep Akata and Samuel Albanie. 2021. Audio Retrieval with Natural Language Queries. In INTERSPEECH. 2411\u20132415.","DOI":"10.21437\/Interspeech.2021-2227"},{"key":"e_1_3_2_1_25_1","unstructured":"Adam Paszke Sam Gross Soumith Chintala Gregory Chanan Edward Yang Zachary DeVito Zeming Lin Alban Desmaison Luca Antiga and Adam Lerer. 2017. Automatic differentiation in pytorch. (2017).  Adam Paszke Sam Gross Soumith Chintala Gregory Chanan Edward Yang Zachary DeVito Zeming Lin Alban Desmaison Luca Antiga and Adam Lerer. 2017. Automatic differentiation in pytorch. (2017)."},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"publisher","DOI":"10.24963\/ijcai.2018\/365"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.1145\/1873951.1873987"},{"key":"e_1_3_2_1_28_1","volume-title":"2002 IEEE International Conference on Acoustics, Speech, and Signal Processing, Vol.\u00a04. IEEE, IV\u20134108","author":"Slaney Malcolm","year":"2002","unstructured":"Malcolm Slaney . 2002 . Semantic-audio retrieval . In 2002 IEEE International Conference on Acoustics, Speech, and Signal Processing, Vol.\u00a04. IEEE, IV\u20134108 . Malcolm Slaney. 2002. Semantic-audio retrieval. In 2002 IEEE International Conference on Acoustics, Speech, and Signal Processing, Vol.\u00a04. IEEE, IV\u20134108."},{"key":"e_1_3_2_1_29_1","doi-asserted-by":"publisher","DOI":"10.1109\/TNN.1998.712192"},{"key":"e_1_3_2_1_30_1","unstructured":"Yusong Wu Ke Chen Tianyu Zhang Yuchen Hui Taylor Berg-Kirkpatrick and Shlomo Dubnov. 2022. Large-scale Contrastive Language-Audio Pretraining with Feature Fusion and Keyword-to-Caption Augmentation. arxiv:2211.06687  Yusong Wu Ke Chen Tianyu Zhang Yuchen Hui Taylor Berg-Kirkpatrick and Shlomo Dubnov. 2022. Large-scale Contrastive Language-Audio Pretraining with Feature Fusion and Keyword-to-Caption Augmentation. arxiv:2211.06687"},{"key":"e_1_3_2_1_31_1","unstructured":"Huang Xie Okko R\u00e4s\u00e4nen and Tuomas Virtanen. 2022. On Negative Sampling for Contrastive Audio-Text Retrieval. arxiv:2211.04070  Huang Xie Okko R\u00e4s\u00e4nen and Tuomas Virtanen. 2022. On Negative Sampling for Contrastive Audio-Text Retrieval. arxiv:2211.04070"},{"key":"e_1_3_2_1_32_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.00800"}],"event":{"name":"CACML 2023: 2023 2nd Asia Conference on Algorithms, Computing and Machine Learning","location":"Shanghai China","acronym":"CACML 2023"},"container-title":["Proceedings of the 2023 2nd Asia Conference on Algorithms, Computing and Machine Learning"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3590003.3590056","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3590003.3590056","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T18:09:16Z","timestamp":1750183756000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3590003.3590056"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,3,17]]},"references-count":32,"alternative-id":["10.1145\/3590003.3590056","10.1145\/3590003"],"URL":"https:\/\/doi.org\/10.1145\/3590003.3590056","relation":{},"subject":[],"published":{"date-parts":[[2023,3,17]]},"assertion":[{"value":"2023-05-29","order":2,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}