{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T09:40:01Z","timestamp":1746697201068,"version":"3.40.5"},"publisher-location":"Cham","reference-count":44,"publisher":"Springer Nature Switzerland","isbn-type":[{"type":"print","value":"9783031887161"},{"type":"electronic","value":"9783031887178"}],"license":[{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,1,1]],"date-time":"2025-01-01T00:00:00Z","timestamp":1735689600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025]]},"DOI":"10.1007\/978-3-031-88717-8_2","type":"book-chapter","created":{"date-parts":[[2025,4,4]],"date-time":"2025-04-04T12:06:51Z","timestamp":1743768411000},"page":"10-20","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["CLASP: Contrastive Language-Speech Pretraining for\u00a0Multilingual Multimodal Information Retrieval"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-1360-6701","authenticated-orcid":false,"given":"Mohammad Mahdi","family":"Abootorabi","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6518-7238","authenticated-orcid":false,"given":"Ehsaneddin","family":"Asgari","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,4,3]]},"reference":[{"key":"2_CR1","first-page":"23716","volume":"35","author":"JB Alayrac","year":"2022","unstructured":"Alayrac, J.B., et al.: Flamingo: a visual language model for few-shot learning. Adv. Neural. Inf. Process. Syst. 35, 23716\u201323736 (2022)","journal-title":"Adv. Neural. Inf. Process. Syst."},{"key":"2_CR2","unstructured":"Ardila, R., et al.: Common voice: a massively-multilingual speech corpus. In: Proceedings of the 12th Conference on Language Resources and Evaluation (LREC 2020), pp. 4211\u20134215 (2020)"},{"key":"2_CR3","unstructured":"Baevski, A., Zhou, Y., Mohamed, A., Auli, M.: wav2vec 2.0: a framework for self-supervised learning of speech representations. In: Advances in Neural Information Processing Systems, vol. 33, pp. 12449\u201312460 (2020)"},{"key":"2_CR4","doi-asserted-by":"crossref","unstructured":"Banerjee, S., Chakravarthi, B.R., McCrae, J.P.: Comparison of pretrained embeddings to identify hate speech in Indian code-mixed text. In: 2020 2nd International Conference on Advances in Computing, Communication Control and Networking (ICACCCN), pp. 21\u201325. IEEE (2020)","DOI":"10.1109\/ICACCCN51052.2020.9362731"},{"key":"2_CR5","doi-asserted-by":"crossref","unstructured":"Cai, X., Yuan, J., Zheng, R., Huang, L., Church, K.: Speech emotion recognition with multi-task learning. In: Interspeech, vol.\u00a02021, pp. 4508\u20134512 (2021)","DOI":"10.21437\/Interspeech.2021-1852"},{"key":"2_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2020.101155","volume":"66","author":"R Cattoni","year":"2021","unstructured":"Cattoni, R., Di Gangi, M.A., Bentivogli, L., Negri, M., Turchi, M.: MuST-C: a multilingual corpus for end-to-end speech translation. Comput. Speech Lang. 66, 101155 (2021)","journal-title":"Comput. Speech Lang."},{"key":"2_CR7","doi-asserted-by":"publisher","unstructured":"Chen, W., Hu, H., Chen, X., Verga, P., Cohen, W.: MuRAG: multimodal retrieval-augmented generator for open question answering over images and text. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 5558\u20135570. Association for Computational Linguistics, Abu Dhabi, United Arab Emirates (2022). https:\/\/doi.org\/10.18653\/v1\/2022.emnlp-main.375, https:\/\/aclanthology.org\/2022.emnlp-main.375","DOI":"10.18653\/v1\/2022.emnlp-main.375"},{"key":"2_CR8","doi-asserted-by":"publisher","unstructured":"Chen, Z., et al.: MAESTRO: matched speech text representations through modality matching. In: Proceedings of Interspeech 2022, pp. 4093\u20134097 (2022). https:\/\/doi.org\/10.21437\/Interspeech.2022-10937","DOI":"10.21437\/Interspeech.2022-10937"},{"key":"2_CR9","doi-asserted-by":"publisher","unstructured":"Chuang, Y.S., Liu, C.L., yi\u00a0Lee, H., shan Lee, L.: SpeechBERT: an audio-and-text jointly learned language model for end-to-end spoken question answering. In: Proceedings of Interspeech 2020, pp. 4168\u20134172 (2020). https:\/\/doi.org\/10.21437\/Interspeech.2020-1570","DOI":"10.21437\/Interspeech.2020-1570"},{"key":"2_CR10","unstructured":"Chung, Y.A., Weng, W.H., Tong, S., Glass, J.: Unsupervised cross-modal alignment of speech and text embedding spaces. In: Bengio, S., Wallach, H., Larochelle, H., Grauman, K., Cesa-Bianchi, N., Garnett, R. (eds.) Advances in Neural Information Processing Systems, vol.\u00a031. Curran Associates, Inc. (2018). https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2018\/file\/1ea97de85eb634d580161c603422437f-Paper.pdf"},{"key":"2_CR11","doi-asserted-by":"publisher","unstructured":"Conneau, A., et al.: Unsupervised cross-lingual representation learning at scale. In: Proceedings of the 58th Annual Meeting of the Association for Computational Linguistics, pp. 8440\u20138451. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.acl-main.747, https:\/\/aclanthology.org\/2020.acl-main.747","DOI":"10.18653\/v1\/2020.acl-main.747"},{"key":"2_CR12","doi-asserted-by":"publisher","unstructured":"Conneau, A., et al.: Fleurs: few-shot learning evaluation of universal representations of speech. In: 2022 IEEE Spoken Language Technology Workshop (SLT), pp. 798\u2013805 (2023). https:\/\/doi.org\/10.1109\/SLT54892.2023.10023141","DOI":"10.1109\/SLT54892.2023.10023141"},{"key":"2_CR13","unstructured":"Elliott, D., Kiela, D., Lazaridou, A.: Multimodal learning and reasoning. In: Proceedings of the 54th Annual Meeting of the Association for Computational Linguistics: Tutorial Abstracts. Association for Computational Linguistics, Berlin, Germany (2016). https:\/\/aclanthology.org\/P16-5001"},{"key":"2_CR14","doi-asserted-by":"publisher","unstructured":"Feng, F., Yang, Y., Cer, D., Arivazhagan, N., Wang, W.: Language-agnostic BERT sentence embedding. In: Proceedings of the 60th Annual Meeting of the Association for Computational Linguistics (Volume 1: Long Papers), pp. 878\u2013891. Association for Computational Linguistics, Dublin, Ireland (2022). https:\/\/doi.org\/10.18653\/v1\/2022.acl-long.62, https:\/\/aclanthology.org\/2022.acl-long.62","DOI":"10.18653\/v1\/2022.acl-long.62"},{"issue":"2","key":"2_CR15","first-page":"7","volume":"5","author":"WN Francis","year":"1979","unstructured":"Francis, W.N., Kucera, H.: Brown corpus manual. Lett. Editor 5(2), 7 (1979)","journal-title":"Lett. Editor"},{"key":"2_CR16","doi-asserted-by":"publisher","unstructured":"Hessel, J., Lee, L.: Does my multimodal model learn cross-modal interactions? It\u2019s harder to tell than you might think! In: Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing (EMNLP), pp. 861\u2013877. Association for Computational Linguistics, Online (2020). https:\/\/doi.org\/10.18653\/v1\/2020.emnlp-main.62, https:\/\/aclanthology.org\/2020.emnlp-main.62","DOI":"10.18653\/v1\/2020.emnlp-main.62"},{"key":"2_CR17","doi-asserted-by":"publisher","unstructured":"Hochreiter, S., Schmidhuber, J.: Long short-term memory. Neural Comput. 9(8), 1735\u20131780 (1997). https:\/\/doi.org\/10.1162\/neco.1997.9.8.1735","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"2_CR18","doi-asserted-by":"publisher","unstructured":"Hsu, W.N., Bolte, B., Tsai, Y.H.H., Lakhotia, K., Salakhutdinov, R., Mohamed, A.: HuBERT: self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Trans. Audio, Speech Lang. Proc. 29, 3451\u20133460 (2021). https:\/\/doi.org\/10.1109\/TASLP.2021.3122291","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"2_CR19","doi-asserted-by":"publisher","unstructured":"Huang, P.Y., Patrick, M., Hu, J., Neubig, G., Metze, F., Hauptmann, A.: Multilingual multimodal pre-training for zero-shot cross-lingual transfer of vision-language models. In: Proceedings of the 2021 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 2443\u20132459. Association for Computational Linguistics, Online (2021). https:\/\/doi.org\/10.18653\/v1\/2021.naacl-main.195, https:\/\/aclanthology.org\/2021.naacl-main.195","DOI":"10.18653\/v1\/2021.naacl-main.195"},{"key":"2_CR20","unstructured":"Jeon, W.: Acoustic neighbor embeddings (2022). https:\/\/arxiv.org\/abs\/2007.10329"},{"key":"2_CR21","unstructured":"Karakos, D., Zbib, R., Hartmann, W., Schwartz, R., Makhoul, J.: Reformulating information retrieval from speech and text as a detection problem. In: Proceedings of the workshop on Cross-Language Search and Summarization of Text and Speech (CLSSTS2020), pp. 38\u201343. European Language Resources Association, Marseille, France (2020). https:\/\/aclanthology.org\/2020.clssts-1.7"},{"issue":"6","key":"2_CR22","doi-asserted-by":"publisher","first-page":"1493","DOI":"10.1109\/JSTSP.2022.3192714","volume":"16","author":"S Khurana","year":"2022","unstructured":"Khurana, S., Laurent, A., Glass, J.: SAMU-XLSR: semantically-aligned multimodal utterance-level cross-lingual speech representation. IEEE J. Sel. Top. Signal Process. 16(6), 1493\u20131504 (2022). https:\/\/doi.org\/10.1109\/JSTSP.2022.3192714","journal-title":"IEEE J. Sel. Top. Signal Process."},{"key":"2_CR23","unstructured":"Kingma, D.P., Ba, J.: Adam: a method for stochastic optimization. CoRR abs\/1412.6980 (2014). https:\/\/api.semanticscholar.org\/CorpusID:6628106"},{"key":"2_CR24","unstructured":"Lakhotia, K., et al.: On generative spoken language modeling from raw audio. Trans. Assoc. Comput. Linguist. 9, 1336\u20131354 (2021)"},{"key":"2_CR25","doi-asserted-by":"crossref","unstructured":"Liu, S., et al.: Audio self-supervised learning: a survey. Patterns 3(12) (2022)","DOI":"10.1016\/j.patter.2022.100616"},{"key":"2_CR26","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1016\/j.specom.2022.02.006","volume":"139","author":"Y Liu","year":"2022","unstructured":"Liu, Y., Sun, H., Guan, W., Xia, Y., Zhao, Z.: Multi-modal speech emotion recognition using self-attention mechanism and multi-scale fusion framework. Speech Commun. 139, 1\u20139 (2022)","journal-title":"Speech Commun."},{"key":"2_CR27","unstructured":"van\u00a0der Maaten, L., Hinton, G.: Visualizing data using t-SNE. J. Mach. Learn. Res. 9, 2579\u20132605 (2008). http:\/\/www.jmlr.org\/papers\/v9\/vandermaaten08a.html"},{"key":"2_CR28","doi-asserted-by":"crossref","unstructured":"Manakul, P., Gales, M.J., Wang, L.: Abstractive spoken document summarization using hierarchical model with multi-stage attention diversity optimization. ISCA (2020)","DOI":"10.21437\/Interspeech.2020-1683"},{"key":"2_CR29","doi-asserted-by":"crossref","unstructured":"Mohamed, A., et\u00a0al.: Self-supervised speech representation learning: a review. IEEE J. Sel. Top. Signal Process. (2022)","DOI":"10.1109\/JSTSP.2022.3207050"},{"key":"2_CR30","doi-asserted-by":"publisher","unstructured":"Mohammadshahi, A., Nikoulina, V., Berard, A., Brun, C., Henderson, J., Besacier, L.: SMaLL-100: introducing shallow multilingual machine translation model for low-resource languages. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 8348\u20138359. Association for Computational Linguistics, Abu Dhabi, United Arab Emirates (2022). https:\/\/doi.org\/10.18653\/v1\/2022.emnlp-main.571, https:\/\/aclanthology.org\/2022.emnlp-main.571","DOI":"10.18653\/v1\/2022.emnlp-main.571"},{"key":"2_CR31","doi-asserted-by":"crossref","unstructured":"Morency, L.P., Baltru\u0161aitis, T.: Multimodal machine learning: Integrating language, vision and speech. In: Proceedings of the 55th Annual Meeting of the Association for Computational Linguistics: Tutorial Abstracts, pp.\u00a03\u20135. Association for Computational Linguistics, Vancouver, Canada (2017). https:\/\/aclanthology.org\/P17-5002","DOI":"10.18653\/v1\/P17-5002"},{"key":"2_CR32","doi-asserted-by":"crossref","unstructured":"Ni, J., et al.: Adaptive knowledge distillation between text and speech pre-trained models. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10096950"},{"key":"2_CR33","unstructured":"Radford, A., et al.: Learning transferable visual models from natural language supervision. In: Meila, M., Zhang, T. (eds.) Proceedings of the 38th International Conference on Machine Learning, ICML 2021, 18-24 July 2021, Virtual Event. Proceedings of Machine Learning Research, vol.\u00a0139, pp. 8748\u20138763. PMLR (2021). http:\/\/proceedings.mlr.press\/v139\/radford21a.html"},{"key":"2_CR34","doi-asserted-by":"crossref","unstructured":"Runz, M., et\u00a0al.: FroDo: from detections to 3D objects. In: Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition, pp. 14720\u201314729 (2020)","DOI":"10.1109\/CVPR42600.2020.01473"},{"key":"2_CR35","doi-asserted-by":"crossref","unstructured":"Schubert, L., Tong, M.: Extracting and evaluating general world knowledge from the brown corpus. In: Proceedings of the HLT-NAACL 2003 Workshop on Text Meaning, pp. 7\u201313 (2003). https:\/\/aclanthology.org\/W03-0902","DOI":"10.3115\/1119239.1119241"},{"key":"2_CR36","doi-asserted-by":"publisher","unstructured":"Shen, J., et al.: Natural TTS synthesis by conditioning WaveNet on MEL spectrogram predictions. In: 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 4779\u20134783 (2018). https:\/\/doi.org\/10.1109\/ICASSP.2018.8461368","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"2_CR37","doi-asserted-by":"publisher","unstructured":"Shih, Y.J., Wang, H.F., Chang, H.J., Berry, L., Lee, H.y., Harwath, D.: SpeechCLIP: integrating speech with pre-trained vision and language model. In: 2022 IEEE Spoken Language Technology Workshop (SLT), pp. 715\u2013722 (2023). https:\/\/doi.org\/10.1109\/SLT54892.2023.10022954","DOI":"10.1109\/SLT54892.2023.10022954"},{"key":"2_CR38","unstructured":"Tan, M., Le, Q.: EfficientNet: rethinking model scaling for convolutional neural networks. In: Chaudhuri, K., Salakhutdinov, R. (eds.) Proceedings of the 36th International Conference on Machine Learning. Proceedings of Machine Learning Research, vol.\u00a097, pp. 6105\u20136114. PMLR (2019)"},{"key":"2_CR39","doi-asserted-by":"publisher","unstructured":"Wang, W., et al.: Optimizing alignment of speech and language latent spaces for end-to-end speech recognition and understanding. In: ICASSP 2022 - 2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7802\u20137806 (2022). https:\/\/doi.org\/10.1109\/ICASSP43922.2022.9747760","DOI":"10.1109\/ICASSP43922.2022.9747760"},{"key":"2_CR40","doi-asserted-by":"crossref","unstructured":"Wang, X., et al.: Large-scale multi-modal pre-trained models: a comprehensive survey (2023)","DOI":"10.1007\/s11633-022-1410-8"},{"key":"2_CR41","doi-asserted-by":"crossref","unstructured":"Xiong, J., Zhou, Y., Zhang, P., Xie, L., Huang, W., Zha, Y.: Look &listen: multi-modal correlation learning for active speaker detection and speech enhancement. IEEE Trans. Multimedia (2022)","DOI":"10.1109\/TMM.2022.3199109"},{"key":"2_CR42","doi-asserted-by":"publisher","unstructured":"Yang, X., et al.: Few-shot joint multimodal aspect-sentiment analysis based on generative multimodal prompt. In: Findings of the Association for Computational Linguistics: ACL 2023, pp. 11575\u201311589. Association for Computational Linguistics, Toronto, Canada (2023). https:\/\/doi.org\/10.18653\/v1\/2023.findings-acl.735, https:\/\/aclanthology.org\/2023.findings-acl.735","DOI":"10.18653\/v1\/2023.findings-acl.735"},{"key":"2_CR43","doi-asserted-by":"publisher","unstructured":"Ye, R., Wang, M., Li, L.: Cross-modal contrastive learning for speech translation. In: Proceedings of the 2022 Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies, pp. 5099\u20135113. Association for Computational Linguistics, Seattle, United States (2022). https:\/\/doi.org\/10.18653\/v1\/2022.naacl-main.376, https:\/\/aclanthology.org\/2022.naacl-main.376","DOI":"10.18653\/v1\/2022.naacl-main.376"},{"key":"2_CR44","doi-asserted-by":"publisher","unstructured":"Zhang, Z., Zhou, L., Ao, J., Liu, S., Dai, L., Li, J., Wei, F.: SpeechUT: bridging speech and text with hidden-unit for encoder-decoder based speech-text pre-training. In: Proceedings of the 2022 Conference on Empirical Methods in Natural Language Processing, pp. 1663\u20131676. Association for Computational Linguistics, Abu Dhabi, United Arab Emirates (2022). https:\/\/doi.org\/10.18653\/v1\/2022.emnlp-main.108, https:\/\/aclanthology.org\/2022.emnlp-main.108","DOI":"10.18653\/v1\/2022.emnlp-main.108"}],"container-title":["Lecture Notes in Computer Science","Advances in Information Retrieval"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-3-031-88717-8_2","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T09:22:53Z","timestamp":1746696173000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-3-031-88717-8_2"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025]]},"ISBN":["9783031887161","9783031887178"],"references-count":44,"URL":"https:\/\/doi.org\/10.1007\/978-3-031-88717-8_2","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"type":"print","value":"0302-9743"},{"type":"electronic","value":"1611-3349"}],"subject":[],"published":{"date-parts":[[2025]]},"assertion":[{"value":"3 April 2025","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"ECIR","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"European Conference on Information Retrieval","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Lucca","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Italy","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2025","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"7 April 2025","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"11 April 2025","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"47","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"ecir2025","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/ecir2025.eu\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}