{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,19]],"date-time":"2026-02-19T02:18:05Z","timestamp":1771467485514,"version":"3.50.1"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1109\/asru46091.2019.9003870","type":"proceedings-article","created":{"date-parts":[[2020,2,21]],"date-time":"2020-02-21T07:01:33Z","timestamp":1582268493000},"page":"928-935","source":"Crossref","is-referenced-by-count":19,"title":["Leveraging Language ID in Multilingual End-to-End Speech Recognition"],"prefix":"10.1109","author":[{"given":"Austin","family":"Waters","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Neeraj","family":"Gaur","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Parisa","family":"Haghani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pedro","family":"Moreno","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhongdi","family":"Qu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","article-title":"In-vestigation into bottle-neck features for meeting speech recognition","author":"gr\u00e9zl","year":"2009","journal-title":"Tenth Annual Conference of the International Speech Communication Association"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461802"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953073"},{"key":"ref13","first-page":"4904","article-title":"Multilingual speech recognition with a single end-to-end model","author":"shubham","year":"2018","journal-title":"2018 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref14","first-page":"4749","article-title":"Multi-dialect speech recognition with a single sequence-to-sequence model","author":"bo","year":"2018","journal-title":"2018 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639654"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682674"},{"key":"ref17","doi-asserted-by":"crossref","first-page":"193","DOI":"10.1109\/ASRU.2017.8268935","article-title":"Exploring architectures, data and units for streaming end-to-end speech recognition with rnn-transducer","author":"rao","year":"2017","journal-title":"2017 IEEE Automatic Speech Recognition and Understanding Workshop ASRU 2017"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref19","author":"he","year":"2019","journal-title":"Streaming end-to-end speech recognition for mobile devices"},{"key":"ref4","first-page":"2837","article-title":"Online and linear-time attention by enforcing monotonic alignments","author":"raffel","year":"2017","journal-title":"Proceedings of the 34th International Conference on Machine Learning-Volume 70 JMLR"},{"key":"ref3","first-page":"5067","article-title":"An online sequence-to-sequence model using partial conditioning","author":"navdeep","year":"2016","journal-title":"Advances in neural information processing systems"},{"key":"ref6","article-title":"Language identification and multilingual speech recognition using discriminatively trained acoustic models","author":"niesler","year":"2006","journal-title":"Workshop on Multilingual Speech and Language Processing"},{"key":"ref5","doi-asserted-by":"crossref","first-page":"1298","DOI":"10.21437\/Interspeech.2017-1705","article-title":"Recurrent neural aligner: An encoder-decoder neural network model for sequence to sequence mapping","author":"sak","year":"2017","journal-title":"InterSpeech"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2014.2364559"},{"key":"ref7","article-title":"Integrating language identification to improve multilingual speech recognition","author":"caesar","year":"2012","journal-title":"Tech Rep IDIAP"},{"key":"ref2","author":"graves","year":"2012","journal-title":"Sequence transduction with recurrent neural networks"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404803"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-64680-0_9"},{"key":"ref22","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2017-1510","article-title":"Generation of large-scale simulated utterances in virtual rooms to train deep-neural networks for far-field speech recognition in Google Home","author":"kim","year":"2017","journal-title":"Proc INTERSPEECH"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462201"},{"key":"ref24","first-page":"265","article-title":"Tensorflow: a system for large-scale machine learning","volume":"16","author":"abadi","year":"2016","journal-title":"OSDI"},{"key":"ref23","article-title":"Fast and accurate recurrent neural network acoustic models for speech recognition","author":"hasim","year":"2015","journal-title":"Proceedings of Interspeech"},{"key":"ref26","author":"kingma","year":"2014","journal-title":"Adam A method for stochastic optimization"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080246"}],"event":{"name":"2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","location":"SG, Singapore","start":{"date-parts":[[2019,12,14]]},"end":{"date-parts":[[2019,12,18]]}},"container-title":["2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8985378\/9003727\/09003870.pdf?arnumber=9003870","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T14:51:20Z","timestamp":1658155880000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9003870\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/asru46091.2019.9003870","relation":{},"subject":[],"published":{"date-parts":[[2019,12]]}}}