{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T18:12:14Z","timestamp":1776881534312,"version":"3.51.2"},"reference-count":42,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,2]]},"DOI":"10.1109\/slt61566.2024.10832238","type":"proceedings-article","created":{"date-parts":[[2025,1,16]],"date-time":"2025-01-16T18:31:27Z","timestamp":1737052287000},"page":"1193-1200","source":"Crossref","is-referenced-by-count":1,"title":["Enhancing Low-Resource Spoken Language Identification Via Cross-Modality Retrieval and Cross-Lingual Text-to-Speech Synthesis"],"prefix":"10.1109","author":[{"given":"Min","family":"Ma","sequence":"first","affiliation":[{"name":"Google"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gary","family":"Wang","sequence":"additional","affiliation":[{"name":"Google"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kyle","family":"Kastner","sequence":"additional","affiliation":[{"name":"Google"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Isaac","family":"Caswell","sequence":"additional","affiliation":[{"name":"Google"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Charles","family":"Yoon","sequence":"additional","affiliation":[{"name":"Google"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Andrew","family":"Rosenberg","sequence":"additional","affiliation":[{"name":"Google"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2012.2237151"},{"key":"ref2","first-page":"105","article-title":"Spoken language recognition using xvectors","volume":"2018","author":"Snyder","year":"2018","journal-title":"Odyssey"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462026"},{"key":"ref4","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proceedings of the 34th NeurIPS Conference","author":"Baevski"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688253"},{"key":"ref6","article-title":"mslam: Massively multilingual joint pre-training for speech and text","volume":"abs\/2202.01374","author":"Bapna","year":"2022","journal-title":"ArXiv"},{"key":"ref7","article-title":"Scaling speech technology to 1,000+ languages","author":"Pratap","year":"2023","journal-title":"arXiv preprint arXiv:2305.13516"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10007"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/SLT54892.2023.10023141"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448215"},{"key":"ref11","article-title":"Xtreme-up: A user-centric scarce-data benchmark for underrepresented languages","volume-title":"The 2023 Conference on Empirical Methods in Natural Language Processing","author":"Ruder"},{"key":"ref12","article-title":"Speechstew: Simply mix all available speech recognition data to train one large neural network","author":"Chan","year":"2021","journal-title":"arXiv preprint arXiv:2104.02133"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2014-207"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1020"},{"key":"ref15","first-page":"21","article-title":"Vocal tract length perturbation (vtlp) improves speech recognition","volume-title":"Proc. ICML Workshop on Deep Learning for Audio, Speech and Language","volume":"117","author":"Jaitly"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2473"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178863"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/WNYIPW.2019.8923082"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003990"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/CISP-BMEI51763.2020.9263564"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688218"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383525"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.23919\/APSIPAASC55919.2022.9980253"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ACCESS.2023.3316142"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10448074"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-021-00225-4"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2668"},{"key":"ref29","first-page":"2709","article-title":"YourTTS: Towards Zero-Shot MultiSpeaker TTS and Zero-Shot Voice Conversion for everyone","volume-title":"International Conference on Machine Learning","author":"Casanova"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v36i10.21327"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747667"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-70136-3_93"},{"key":"ref33","article-title":"Ambernet: A compact end-to-end model for spoken language identification","author":"Jia","year":"2022","journal-title":"arXiv preprint arXiv:2210.15781"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-241"},{"key":"ref35","article-title":"Masr: Metadata aware speech representation","author":"Raj","year":"2023","journal-title":"arXiv preprint arXiv:2307.10982"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-854"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.coling-main.579"},{"key":"ref38","article-title":"Corpus Crawler","author":"Brawer","year":"2017"},{"key":"ref39","article-title":"Bilex rx: Lexical data augmentation for massively multilingual machine translation","author":"Jones","year":"2023","journal-title":"arXiv preprint arXiv:2303.15265"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1561\/2200000056"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10094909"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-srw.12"}],"event":{"name":"2024 IEEE Spoken Language Technology Workshop (SLT)","location":"Macao","start":{"date-parts":[[2024,12,2]]},"end":{"date-parts":[[2024,12,5]]}},"container-title":["2024 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10830790\/10830793\/10832238.pdf?arnumber=10832238","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T07:44:13Z","timestamp":1737099853000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10832238\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"references-count":42,"URL":"https:\/\/doi.org\/10.1109\/slt61566.2024.10832238","relation":{},"subject":[],"published":{"date-parts":[[2024,12,2]]}}}