{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,9]],"date-time":"2026-05-09T17:21:18Z","timestamp":1778347278110,"version":"3.51.4"},"reference-count":43,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Top. Signal Process."],"published-print":{"date-parts":[[2022,10]]},"DOI":"10.1109\/jstsp.2022.3192714","type":"journal-article","created":{"date-parts":[[2022,7,20]],"date-time":"2022-07-20T19:36:32Z","timestamp":1658345792000},"page":"1493-1504","source":"Crossref","is-referenced-by-count":20,"title":["SAMU-XLSR: Semantically-Aligned Multimodal Utterance-Level Cross-Lingual Speech Representation"],"prefix":"10.1109","volume":"16","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3182-1085","authenticated-orcid":false,"given":"Sameer","family":"Khurana","sequence":"first","affiliation":[{"name":"MIT Computer Science, Artificial Intelligence Laboratory, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2653-1008","authenticated-orcid":false,"given":"Antoine","family":"Laurent","sequence":"additional","affiliation":[{"name":"LIUM&#x2014;Le Mans University, Le Mans, France"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3097-360X","authenticated-orcid":false,"given":"James","family":"Glass","sequence":"additional","affiliation":[{"name":"MIT Computer Science, Artificial Intelligence Laboratory, Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Advances in Neural Information Processing Systems","volume":"33","author":"Baevski","year":"2020"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054438"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-349"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2605"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3084"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-329"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2022-143"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/jstsp.2022.3188113"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688253"},{"key":"ref12","article-title":"mSLAM: Massively multilingual joint pre-training for speech and text","author":"Bapna","year":"2022"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/W17-2619"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00288"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.62"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-2037"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.eacl-main.115"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.507"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1121"},{"key":"ref20","article-title":"The missing ingredient in zero-shot neural machine translation","author":"Arivazhagan","year":"2019"},{"key":"ref21","first-page":"15748","article-title":"Multimodal and multilingual embeddings for large-scale speech mining","author":"Duquenne","year":"2021","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref22","article-title":"Attention is all you need","volume-title":"Advances in Neural Information Processing Systems","volume":"30","author":"Vaswani","year":"2017"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1446"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref25","article-title":"Cross-lingual language model pretraining","volume-title":"Advances in Neural Information Processing Systems","volume":"32","author":"Conneau","year":"2019"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6289079"},{"key":"ref27","article-title":"Googles neural machine translation system: Bridging the gap between human and machine translation","author":"Wu","year":"2016"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.emnlp-demos.6"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.747"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00343"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref32","article-title":"CoVoST: A diverse multilingual speech-to-text translation corpus","author":"Wang","year":"2020"},{"key":"ref33","article-title":"Unsupervised learning of spoken language with visual context","volume":"29","author":"Harwath","year":"2016","journal-title":"Adv. Neural Inf. Process. Syst."},{"key":"ref34","article-title":"Learning hierarchical discrete linguistic units from visually-grounded speech","volume-title":"Proc. 8th Int. Conf. Learn. Representations","author":"Harwath","year":"2020"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-1312"},{"key":"ref36","article-title":"Word error rate  wikipedia, the free encyclopedia","volume-title":"Wikipedia Contributors","year":"2022"},{"key":"ref37","first-page":"2012","article-title":"MuST-C: A multilingual speech translation corpus","volume-title":"Proc. Conf. North Amer. Chapter Assoc. Comput. Linguistics: Hum. Lang. Technol.","author":"Gangi","year":"2019"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-11"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.80"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2018-1456"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/DSLW51110.2021.9523402"},{"key":"ref43","article-title":"Common voice: A massively-multilingual speech corpus","author":"Ardila","year":"2020"}],"container-title":["IEEE Journal of Selected Topics in Signal Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/4200690\/9923627\/09834099.pdf?arnumber=9834099","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T06:42:50Z","timestamp":1706769770000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9834099\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10]]},"references-count":43,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/jstsp.2022.3192714","relation":{},"ISSN":["1932-4553","1941-0484"],"issn-type":[{"value":"1932-4553","type":"print"},{"value":"1941-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,10]]}}}