{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,21]],"date-time":"2026-02-21T18:17:42Z","timestamp":1771697862244,"version":"3.50.1"},"reference-count":29,"publisher":"IEEE","content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2018,4]]},"DOI":"10.1109\/icassp.2018.8461361","type":"proceedings-article","created":{"date-parts":[[2018,9,21]],"date-time":"2018-09-21T22:24:48Z","timestamp":1537568688000},"page":"4754-4758","source":"Crossref","is-referenced-by-count":21,"title":["On Modular Training of Neural Acoustics-to-Word Model for LVCSR"],"prefix":"10.1109","author":[{"given":"Zhehuai","family":"Chen","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Qi","family":"Liu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hao","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Kai","family":"Yu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2625459"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref13","article-title":"Direct acoustics-to-word models for english conversational speech recognition","author":"audhkhasi","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1162"},{"key":"ref15","doi-asserted-by":"crossref","first-page":"1641","DOI":"10.21437\/Interspeech.2017-1284","article-title":"End-to-end training of acoustic models for large vocabulary continuous speech recognition with tensorflow","author":"variani","year":"2017","journal-title":"Proc Interspeech 2017"},{"key":"ref16","doi-asserted-by":"crossref","first-page":"939","DOI":"10.21437\/Interspeech.2017-233","article-title":"A comparison of sequence-to-sequence models for speech recognition","author":"prabhavalkar","year":"2017","journal-title":"Proc Interspeech 2017"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"ref18","doi-asserted-by":"crossref","first-page":"3702","DOI":"10.21437\/Interspeech.2017-232","article-title":"An analysis of attention in sequence-to-sequence models","author":"prabhavalkar","year":"2017","journal-title":"Proc Interspeech 2017"},{"key":"ref19","doi-asserted-by":"crossref","first-page":"3692","DOI":"10.21437\/Interspeech.2017-751","article-title":"Gaussian prediction based attention for online end-to-end speech recognition","author":"hou","year":"2017","journal-title":"Proc Interspeech 2017"},{"key":"ref28","author":"paszke","year":"2017","journal-title":"PyTorch"},{"key":"ref4","article-title":"Neural speech recognizer: Acoustic-to-word lstm model for large vocabulary speech recognition","author":"soltau","year":"2016","journal-title":"ArXiv Preprint"},{"key":"ref27","first-page":"517","article-title":"Switchboard: Telephone speech corpus for research and development","volume":"1","author":"john","year":"1992","journal-title":"Acoustics Speech and Signal Processing 1992 ICASSP-92 1992 IEEE International Conference on"},{"key":"ref3","article-title":"Fast and accurate recurrent neural network acoustic models for speech recognition","author":"sak","year":"2015","journal-title":"ArXiv Preprint"},{"key":"ref6","doi-asserted-by":"crossref","first-page":"1298","DOI":"10.21437\/Interspeech.2017-1705","article-title":"Recurrent neural aligner: An encoder-decoder neural network model for sequence to sequence mapping","author":"sak","year":"2017","journal-title":"Proc Interspeech 2017"},{"key":"ref29","article-title":"The kaldi speech recognition toolkit","author":"povey","year":"2011","journal-title":"IEEE 2011 Workshop on Automatic Speech Recognition and Understanding IEEE Signal Processing Society"},{"key":"ref5","article-title":"Sequence transduction with recurrent neural networks","author":"graves","year":"2012","journal-title":"ArXiv Preprint"},{"key":"ref8","article-title":"Wav2letter: an end-to-end convnet-based speech recognition system","author":"collobert","year":"2016","journal-title":"ArXiv Preprint"},{"key":"ref7","article-title":"End-to-end neural segmental models for speech recognition","author":"hao","year":"2017","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2602884"},{"key":"ref9","author":"chan","year":"2016","journal-title":"End-to-End Speech Recognition Models"},{"key":"ref1","article-title":"Long short-term memory recurrent neural network architectures for large scale acoustic modeling","author":"sak","year":"2014","journal-title":"Fifteenth Annual Conference of the International Speech Communication Association"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952667"},{"key":"ref22","article-title":"Cold fusion: Training seq2seq models together with language models","author":"sriram","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref21","article-title":"Advances in joint etc-attention based end-to-end speech recognition with a deep cnn encoder and rnn-lm","author":"hori","year":"2017","journal-title":"ArXiv Preprint"},{"key":"ref24","article-title":"Exploring the limits of language modeling","author":"jozefowicz","year":"2016","journal-title":"ArXiv Preprint"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(85)90037-8"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-595"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404790"}],"event":{"name":"ICASSP 2018 - 2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Calgary, AB","start":{"date-parts":[[2018,4,15]]},"end":{"date-parts":[[2018,4,20]]}},"container-title":["2018 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8450881\/8461260\/08461361.pdf?arnumber=8461361","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2020,8,24]],"date-time":"2020-08-24T05:08:10Z","timestamp":1598245690000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8461361\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2018,4]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/icassp.2018.8461361","relation":{},"subject":[],"published":{"date-parts":[[2018,4]]}}}