{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,7]],"date-time":"2026-05-07T17:03:03Z","timestamp":1778173383324,"version":"3.51.4"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2023,8,2]],"date-time":"2023-08-02T00:00:00Z","timestamp":1690934400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2023,8,2]],"date-time":"2023-08-02T00:00:00Z","timestamp":1690934400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"funder":[{"DOI":"10.13039\/501100014879","name":"Central South University of Forestry and Technology","doi-asserted-by":"publisher","award":["CX202202081"],"award-info":[{"award-number":["CX202202081"]}],"id":[{"id":"10.13039\/501100014879","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Multimedia Systems"],"published-print":{"date-parts":[[2023,12]]},"DOI":"10.1007\/s00530-023-01144-4","type":"journal-article","created":{"date-parts":[[2023,8,2]],"date-time":"2023-08-02T11:05:33Z","timestamp":1690974333000},"page":"3625-3638","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":10,"title":["Audio\u2013text retrieval based on contrastive learning and collaborative attention mechanism"],"prefix":"10.1007","volume":"29","author":[{"given":"Tao","family":"Hu","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xuyu","family":"Xiang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaohua","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yun","family":"Tan","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2023,8,2]]},"reference":[{"key":"1144_CR1","doi-asserted-by":"crossref","unstructured":"Jiang, Q.Y., Li, W. J.: Deep cross-modal hashing. In: Proceedings of the IEEE conference on computer vision and pattern recognition. (2017) p. 3232\u20133240.","DOI":"10.1109\/CVPR.2017.348"},{"key":"1144_CR2","doi-asserted-by":"crossref","unstructured":"Li, C., Deng, C., Li, N., et al.: Self-supervised adversarial hashing networks for cross-modal retrieval. Proceedings of the IEEE conference on computer vision and pattern recognition. 4242\u20134251 (2018)","DOI":"10.1109\/CVPR.2018.00446"},{"issue":"4","key":"1144_CR3","doi-asserted-by":"publisher","first-page":"1602","DOI":"10.1109\/TIP.2018.2878970","volume":"28","author":"L Wu","year":"2018","unstructured":"Wu, L., Wang, Y., Shao, L.: Cycle-consistent deep generative hashing for cross-modal retrieval[J]. IEEE Trans. Image Process. 28(4), 1602\u20131612 (2018)","journal-title":"IEEE Trans. Image Process."},{"issue":"1","key":"1144_CR4","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1145\/3281746","volume":"15","author":"Y Yu","year":"2019","unstructured":"Yu, Y., Tang, S., Raposo, F., et al.: Deep cross-modal correlation learning for audio and lyrics in music retrieval. ACM Transact. Multimed. Comput. Commun. Appl. (TOMM) 15(1), 1\u201316 (2019)","journal-title":"ACM Transact. Multimed. Comput. Commun. Appl. (TOMM)"},{"key":"1144_CR5","doi-asserted-by":"crossref","unstructured":"Lou, S., Xu, X., Wu, M., et al.: Audio-Text Retrieval in Context. In: ICASSP 2022\u20132022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2022: 4793\u20134797","DOI":"10.1109\/ICASSP43922.2022.9746786"},{"key":"1144_CR6","unstructured":"Liu, J., Zhu, X., Liu, F., et al.: Opt: omni-perception pre-trainer for cross-modal understanding and generation. https:\/\/arxiv.org\/abs\/2107.00249, (2021)"},{"key":"1144_CR7","unstructured":"Manco, I., Benetos, E., Quinton, E., et al.: Contrastive audio-language learning for music[J]. https:\/\/arxiv.org\/abs\/2208.12208, (2022)"},{"key":"1144_CR8","doi-asserted-by":"crossref","unstructured":"Won, M., Oramas, S., Nieto, O., et al.: Multimodal metric learning for tag-based music retrieval. In: ICASSP 2021\u20132021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2021: 591\u2013595","DOI":"10.1109\/ICASSP39728.2021.9413514"},{"key":"1144_CR9","unstructured":"Won, M., Salamon, J., Bryan, N. J., et al.: Emotion Embedding Spaces for Matching Music to Stories[J]. https:\/\/arxiv.org\/abs\/2111.13468, (2021)"},{"key":"1144_CR10","doi-asserted-by":"crossref","unstructured":"Zhang, Hongli.: \"Voice keyword retrieval method using attention mechanism and multimodal information fusion.\"\u00a0Scientific Programming\u00a02021 (2021)","DOI":"10.1155\/2021\/6662841"},{"key":"1144_CR11","unstructured":"Mei, X., Huang, Q., Liu, X., et al.: An encoder-decoder based audio captioning system with transfer and reinforcement learning. https:\/\/arxiv.org\/abs\/2108.02752, (2021)"},{"key":"1144_CR12","first-page":"229","volume-title":"Audio interval retrieval using convolutional neural networks. Internet of Things, Smart Spaces, and Next Generation Networks and Systems","author":"I Kuzminykh","year":"2020","unstructured":"Kuzminykh, I., Shevchuk, D., Shiaeles, S., et al.: Audio interval retrieval using convolutional neural networks. Internet of Things, Smart Spaces, and Next Generation Networks and Systems, pp. 229\u2013240. Springer, Cham (2020)"},{"key":"1144_CR13","doi-asserted-by":"crossref","unstructured":"Koepke, A. S., Oncescu, A. M., Henriques, J., et al.: Audio retrieval with natural language queries: A benchmark study. IEEE Transact on Multimedia, (2022)","DOI":"10.21437\/Interspeech.2021-2227"},{"issue":"2","key":"1144_CR14","doi-asserted-by":"publisher","first-page":"200","DOI":"10.1007\/s12559-013-9231-2","volume":"6","author":"A Abel","year":"2014","unstructured":"Abel, A., Hussain, A.: Novel two-stage audiovisual speech filtering in noisy environments[J]. Cogn. Comput. 6(2), 200\u2013217 (2014)","journal-title":"Cogn. Comput."},{"issue":"6","key":"1144_CR15","doi-asserted-by":"publisher","first-page":"1642","DOI":"10.1109\/TASL.2010.2096212","volume":"19","author":"I Almajai","year":"2010","unstructured":"Almajai, I., Milner, B.: Visually derived wiener filters for speech enhancement. IEEE Trans. Audio Speech Lang. Process. 19(6), 1642\u20131651 (2010)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"9","key":"1144_CR16","doi-asserted-by":"publisher","first-page":"1900","DOI":"10.1109\/TASL.2013.2261814","volume":"21","author":"MS Khan","year":"2013","unstructured":"Khan, M.S., Naqvi, S.M., Wang, W., et al.: Video-aided model-based source separation in real reverberant rooms. IEEE Trans. Audio Speech Lang. Process. 21(9), 1900\u20131912 (2013)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"1","key":"1144_CR17","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1186\/1687-6180-2012-1","volume":"2012","author":"Y Liang","year":"2012","unstructured":"Liang, Y., Naqvi, S.M., Chambers, J.A.: Audio video based fast fixed-point independent vector analysis for multisource separation in a room environment[J]. EURASIP J. Adv. Sig. Process. 2012(1), 1\u201316 (2012)","journal-title":"EURASIP J. Adv. Sig. Process."},{"issue":"8","key":"1144_CR18","doi-asserted-by":"publisher","first-page":"2257","DOI":"10.1109\/TASL.2007.906197","volume":"15","author":"HK Maganti","year":"2007","unstructured":"Maganti, H.K., Gatica-Perez, D., McCowan, I.: Speech enhancement and recognition in meetings with an audio\u2013visual sensor array[J]. IEEE Trans. Audio Speech Lang. Process. 15(8), 2257\u20132269 (2007)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"issue":"1","key":"1144_CR19","doi-asserted-by":"publisher","first-page":"96","DOI":"10.1109\/TASL.2006.872619","volume":"15","author":"B Rivet","year":"2006","unstructured":"Rivet, B., Girin, L., Jutten, C.: Mixing audiovisual speech processing and blind source separation for the extraction of speech signals from convolutive mixtures. IEEE Trans. Audio Speech Lang. Process. 15(1), 96\u2013108 (2006)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"1144_CR20","doi-asserted-by":"publisher","first-page":"1899","DOI":"10.1109\/TSP.2021.3066038","volume":"69","author":"M Sadeghi","year":"2021","unstructured":"Sadeghi, M., Alameda-Pineda, X.: Mixture of inference networks for VAE-based audio-visual speech enhancement. IEEE Trans. Signal Process. 69, 1899\u20131909 (2021)","journal-title":"IEEE Trans. Signal Process."},{"key":"1144_CR21","doi-asserted-by":"crossref","unstructured":"Sadeghi, M., Alameda-Pineda, X.: Robust unsupervised audio-visual speech enhancement using a mixture of variational autoencoders. In: ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2020: 7534\u20137538","DOI":"10.1109\/ICASSP40776.2020.9053730"},{"key":"1144_CR22","unstructured":"Ideli, E.: Audio-visual speech processing using deep learning techniques. Applied Sciences: School of Engineering Science, (2019)"},{"key":"1144_CR23","doi-asserted-by":"crossref","unstructured":"Ideli, E., Sharpe, B., Baji\u0107, I. V., et al.: Visually assisted time-domain speech enhancement. 2019 IEEE global conference on signal and information processing (GlobalSIP). IEEE, 2019: 1\u20135","DOI":"10.1109\/GlobalSIP45357.2019.8969244"},{"issue":"3","key":"1144_CR24","doi-asserted-by":"publisher","first-page":"589","DOI":"10.1007\/s12559-019-09653-z","volume":"12","author":"A Adeel","year":"2020","unstructured":"Adeel, A., Ahmad, J., Larijani, H., et al.: A novel real-time, lightweight chaotic-encryption scheme for next-generation audio-visual hearing aids. Cogn. Comput. 12(3), 589\u2013601 (2020)","journal-title":"Cogn. Comput."},{"key":"1144_CR25","unstructured":"Adeel, A., Gogate, M., Hussain, A.: Towards next-generation lipreading driven hearing-aids: A preliminary prototype demo[C]\/\/Proceedings of the International Workshop on Challenges in Hearing Assistive Technology (CHAT-2017), Stockholm, Sweden. 2017, 19"},{"key":"1144_CR26","doi-asserted-by":"crossref","unstructured":"Afouras, T., Chung, J. S., Zisserman. A.: My lips are concealed: Audio-visual speech enhancement through obstructions[J]. https:\/\/arxiv.org\/abs\/1907.04975, (2019)","DOI":"10.21437\/Interspeech.2019-3114"},{"key":"1144_CR27","doi-asserted-by":"crossref","unstructured":"Arriandiaga, A., Morrone, G., Pasa, L., et al.: Audio-visual target speaker enhancement on multi-talker environment using event-driven cameras. In: 2021 IEEE International Symposium on Circuits and Systems (ISCAS). IEEE, 2021: 1\u20135","DOI":"10.1109\/ISCAS51556.2021.9401772"},{"key":"1144_CR28","doi-asserted-by":"crossref","unstructured":"Wu, Z., Xiong, Y., Yu, S. X., et al.: Unsupervised feature learning via non-parametric instance discrimination. In: Proceedings of the IEEE conference on computer vision and pattern recognition. 2018: 3733\u20133742","DOI":"10.1109\/CVPR.2018.00393"},{"key":"1144_CR29","doi-asserted-by":"crossref","unstructured":"Ye, M., Zhang, X., Yuen, P. C., et al.: Unsupervised embedding learning via invariant and spreading instance feature. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2019: 6210\u20136219","DOI":"10.1109\/CVPR.2019.00637"},{"key":"1144_CR30","unstructured":"Oord, A., Li, Y., Vinyals, O.: Representation learning with contrastive predictive coding[J]. https:\/\/arxiv.org\/abs\/1807.03748, (2018)"},{"key":"1144_CR31","doi-asserted-by":"crossref","unstructured":"Tian, Y., Krishnan, D., Isola, P.: Contrastive multiview coding. In: European conference on computer vision. Springer, Cham, 2020: 776\u2013794","DOI":"10.1007\/978-3-030-58621-8_45"},{"key":"1144_CR32","unstructured":"Jia, C., Yang, Y., Xia, Y., et al.: Scaling up visual and vision-language representation learning with noisy text supervision. In: International Conference on Machine Learning. PMLR, 2021: 4904\u20134916."},{"key":"1144_CR33","first-page":"9694","volume":"34","author":"J Li","year":"2021","unstructured":"Li, J., Selvaraju, R., Gotmare, A., et al.: Align before fuse: Vision and language representation learning with momentum distillation. Adv. Neural. Inf. Process. Syst. 34, 9694\u20139705 (2021)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"1144_CR34","unstructured":"Wang, W., Bao, H., Dong, L., et al.: Vlmo: Unified vision-language pre-training with mixture-of-modality-experts[J]. https:\/\/arxiv.org\/abs\/2111.02358, (2021)"},{"key":"1144_CR35","unstructured":"Shen, D., Zheng, M., Shen, Y., et al.: A simple but tough-to-beat data augmentation approach for natural language understanding and generation. arXiv preprint arXiv:2009.13818, (2020)"},{"key":"1144_CR36","doi-asserted-by":"crossref","unstructured":"Fang, H., Wang, S., Zhou, M., et al.: Cert: Contrastive self-supervised learning for language understanding. https:\/\/arxiv.org\/abs\/2005.12766, (2020)","DOI":"10.36227\/techrxiv.12308378.v1"},{"key":"1144_CR37","unstructured":"Wu, X., Gao, C., Zang, L., et al.: Esimcse: Enhanced sample building method for contrastive learning of unsupervised sentence embedding[J]. https:\/\/arxiv.org\/abs\/2109.04380, (2021)"},{"key":"1144_CR38","unstructured":"Li, W., Gao, C., Niu, G., et al.: Unimo: Towards unified-modal understanding and generation via cross-modal contrastive learning[J]. https:\/\/arxiv.org\/abs\/2012.15409, (2020)"},{"key":"1144_CR39","doi-asserted-by":"crossref","unstructured":"Zhang, H., Koh, J. Y., Baldridge, J., et al.: Cross-modal contrastive learning for text-to-image generation. Proceedings of the IEEE\/CVF conference on computer vision and pattern recognition. 2021: 833\u2013842","DOI":"10.1109\/CVPR46437.2021.00089"},{"key":"1144_CR40","unstructured":"Liu, J., Zhu, X., Liu, F., et al.: OPT: Omni-perception pre-trainer for cross-modal understanding and generation[J]. arXiv preprint arXiv:2107.00249, (2021)"},{"key":"1144_CR41","doi-asserted-by":"crossref","unstructured":"Seo, P. H., Nagrani, A., Arnab, A., et al:. End-to-end generative pretraining for multimodal video captioning. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2022: 17959\u201317968","DOI":"10.1109\/CVPR52688.2022.01743"},{"key":"1144_CR42","unstructured":"Guu, K., Lee, K., Tung, Z., et al.: REALM: Retrieval-Augmented Language Model Pre[J]. Training, 2020."},{"key":"1144_CR43","doi-asserted-by":"crossref","unstructured":"Mei, X., Liu, X., Sun, J., et al.: On Metric Learning for Audio-Text Cross-Modal Retrieval[J]. https:\/\/arxiv.org\/abs\/2203.15537, (2022)","DOI":"10.21437\/Interspeech.2022-11115"},{"key":"1144_CR44","unstructured":"Chen, T., Kornblith, S., Norouzi, M., et al.: A simple framework for contrastive learning of visual representations. In: International conference on machine learning. PMLR, 2020: 1597\u20131607"},{"key":"1144_CR45","unstructured":"Kim, C. D., Kim, B., Lee, H., et al.: Audiocaps: Generating captions for audios in the wild. In: Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, Volume 1 (Long and Short Papers). 2019: 119\u2013132"},{"key":"1144_CR46","doi-asserted-by":"crossref","unstructured":"Drossos, K., Lipping, S., Virtanen, T.: Clotho: An audio captioning dataset[C]\/\/ICASSP 2020\u20132020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). IEEE, 2020: 736\u2013740","DOI":"10.1109\/ICASSP40776.2020.9052990"},{"key":"1144_CR47","doi-asserted-by":"crossref","unstructured":"Bogolin, S. V., Croitoru. I., Jin, H., et al.: Cross modal retrieval with querybank normalization. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition. 2022: 5194\u20135205.","DOI":"10.1109\/CVPR52688.2022.00513"}],"container-title":["Multimedia Systems"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-023-01144-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00530-023-01144-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00530-023-01144-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,16]],"date-time":"2023-11-16T06:06:37Z","timestamp":1700114797000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00530-023-01144-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,8,2]]},"references-count":47,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2023,12]]}},"alternative-id":["1144"],"URL":"https:\/\/doi.org\/10.1007\/s00530-023-01144-4","relation":{"has-preprint":[{"id-type":"doi","id":"10.21203\/rs.3.rs-2371994\/v1","asserted-by":"object"}]},"ISSN":["0942-4962","1432-1882"],"issn-type":[{"value":"0942-4962","type":"print"},{"value":"1432-1882","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023,8,2]]},"assertion":[{"value":"13 December 2022","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"17 July 2023","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"2 August 2023","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no competing interests.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}