{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,1,9]],"date-time":"2026-01-09T21:12:44Z","timestamp":1767993164755,"version":"3.49.0"},"reference-count":40,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1109\/asru46091.2019.9003730","type":"proceedings-article","created":{"date-parts":[[2020,2,21]],"date-time":"2020-02-21T07:01:33Z","timestamp":1582268493000},"page":"54-61","source":"Crossref","is-referenced-by-count":38,"title":["State-of-the-Art Speech Recognition Using Multi-Stream Self-Attention with Dilated 1D Convolutions"],"prefix":"10.1109","author":[{"given":"Kyu J.","family":"Han","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ramon","family":"Prieto","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tao","family":"Ma","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2460"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2064307"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1006\/csla.1998.0043"},{"key":"ref31","article-title":"The Kaldi speech recognition toolkit","author":"povey","year":"2011","journal-title":"Automatic Speech Recognition and Understanding Workshop"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461974"},{"key":"ref37","author":"luscher","year":"2019","journal-title":"RWTH ASR systems for LibriSpeech Hybrid vs attention - w\/o data augmentation"},{"key":"ref36","author":"yang","year":"2018","journal-title":"A novel pyramidal-FSMN architecture with lattice-free MMI for speech recognition"},{"key":"ref35","author":"chandrashekaran","year":"2018","journal-title":"The CAPIO 2017 Conversational Speech Recognition System"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-595"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1107"},{"key":"ref40","author":"karita","year":"2019","journal-title":"A comprative study on Transformer vs RNN in speech applications"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref13","first-page":"5206","article-title":"LibrSspeech: An ASR corpus based on public domain audio books","author":"panayotov","year":"2015","journal-title":"International Conference on Acoustics Speech and Signal Processing"},{"key":"ref14","author":"ba","year":"2016","journal-title":"Layer normalization"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/29.21701"},{"key":"ref16","first-page":"3214","article-title":"A time delay neural network architecture for efficient modeling of long temporal contexts","author":"peddinti","year":"2015","journal-title":"INTER-SPEECH"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1417"},{"key":"ref18","first-page":"2365","article-title":"Restructuring of deep neural network acoustic models with singular value decomposition","author":"xue","year":"2013","journal-title":"InterSpeech"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472823"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1995.479394"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"ref27","first-page":"901","article-title":"SRILM - An extensible language modeling toolkit","author":"stolcke","year":"2002","journal-title":"International Conference on Spoken Language Processing"},{"key":"ref3","author":"yang","year":"2019","journal-title":"XL-Net Generalized autoregressive pretraining for language understanding"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462105"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.3115\/981863.981904"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462497"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682539"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1910"},{"key":"ref2","author":"devlin","year":"2018","journal-title":"BERT Pre-training of deep bidirectional transformers for language understanding"},{"key":"ref9","first-page":"5884","author":"dong","year":"2018","journal-title":"Speech-Transformer A no-recurrence sequence-to-sequence model for speech recognition"},{"key":"ref1","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Neural Information Processing Systems"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1393"},{"key":"ref22","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"2015","journal-title":"International Conference on Machine Learning"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638312"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref23","author":"hinton","year":"2012","journal-title":"Improving Neural Networks by Preventing Co-adaptation of Feature Detectors"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2008.01.002"},{"key":"ref25","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1108\/RR-08-2013-0197","article-title":"LibriVox: Free public domain audio-books","volume":"28","author":"kearns","year":"2014","journal-title":"Reference Reviews"}],"event":{"name":"2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","location":"SG, Singapore","start":{"date-parts":[[2019,12,14]]},"end":{"date-parts":[[2019,12,18]]}},"container-title":["2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8985378\/9003727\/09003730.pdf?arnumber=9003730","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,18]],"date-time":"2022-07-18T14:51:19Z","timestamp":1658155879000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9003730\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12]]},"references-count":40,"URL":"https:\/\/doi.org\/10.1109\/asru46091.2019.9003730","relation":{},"subject":[],"published":{"date-parts":[[2019,12]]}}}