{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,3,24]],"date-time":"2025-03-24T08:14:10Z","timestamp":1742804050583},"reference-count":31,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,5]]},"DOI":"10.1109\/icassp.2019.8683843","type":"proceedings-article","created":{"date-parts":[[2019,4,17]],"date-time":"2019-04-17T20:01:56Z","timestamp":1555531316000},"page":"5661-5665","source":"Crossref","is-referenced-by-count":13,"title":["Large Context End-to-end Automatic Speech Recognition via Extension of Hierarchical Recurrent Encoder-decoder Models"],"prefix":"10.1109","author":[{"given":"Ryo","family":"Masumura","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tomohiro","family":"Tanaka","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takafumi","family":"Moriya","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yusuke","family":"Shinohara","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takanobu","family":"Oba","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yushi","family":"Aono","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1166"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2400218"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462105"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1616"},{"key":"ref12","first-page":"193","article-title":"Exploring architectures, data and units for streaming end-to-end speech recognition with RNN-transducer","author":"rao","year":"2017","journal-title":"Proc Automatic Speech Recognition and Understanding Workshop (ASRU)"},{"key":"ref13","first-page":"1298","article-title":"Recurrent neural aligner: An encoder-decoder neural network model for sequence to sequence mapping","author":"sak","year":"2017","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1106"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1125"},{"key":"ref16","first-page":"5715","article-title":"Dialogue context language modeling with recurrent neural networks","author":"liu","year":"2017","journal-title":"Proc International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2185"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/2806416.2806493"},{"key":"ref19","first-page":"3776","article-title":"Building end-to-end dialogue systems using generative hierarchical neural network models","author":"serban","year":"2016","journal-title":"Proceedings of the AAAI National Conference on Artificial Intelligence (AAAI)"},{"key":"ref28","first-page":"949","article-title":"Advances in joint CTC-attention based end-to-end speech recognition with a deep CNN encoder and RNN-LM","author":"hori","year":"2017","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref4","first-page":"3707","article-title":"Neural speech recognizer: Acoustic-to-word LSTM model for large vocabulary speech recognition","author":"soltau","year":"2017","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref27","first-page":"947","article-title":"Spontaneous speech corpus of Japanese","author":"maekawa","year":"2000","journal-title":"Proc Int Conference on Language Resources and Evaluation (LREC)"},{"key":"ref3","first-page":"959","article-title":"Direct acoustics-to-word models for English conversational speech recognition","author":"audhkhasi","year":"2017","journal-title":"Conference of the International Speech Communication Association (Inter-Speech)"},{"key":"ref6","first-page":"4945","article-title":"End-to-end attention-based large vocabulary speech recognition","author":"bahdanau","year":"2015","journal-title":"Proc International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref29","first-page":"1045","article-title":"Recurrent neural network based language model","author":"mikolov","year":"2010","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461935"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref7","first-page":"3249","article-title":"A study of the recurrent neural network encoder-decoder for large vocabulary speech recognition","author":"lu","year":"2015","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953069"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472641"},{"key":"ref1","first-page":"1468","article-title":"Fast and accurate recurrent neural network acoustic models for speech recognition","author":"sak","year":"2015","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref20","first-page":"3295","article-title":"A hierarchical latent variable encoder-decoder model for generating dialogues","author":"serban","year":"2017","journal-title":"Proceedings of the AAAI National Conference on Artificial Intelligence (AAAI)"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268945"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1438"},{"article-title":"Does neural machine translation benefit from larger context?","year":"2017","author":"jean","key":"ref24"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461886"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1118"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1301"}],"event":{"name":"ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2019,5,12]]},"location":"Brighton, United Kingdom","end":{"date-parts":[[2019,5,17]]}},"container-title":["ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8671773\/8682151\/08683843.pdf?arnumber=8683843","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,15]],"date-time":"2022-07-15T03:14:13Z","timestamp":1657854853000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8683843\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5]]},"references-count":31,"URL":"https:\/\/doi.org\/10.1109\/icassp.2019.8683843","relation":{},"subject":[],"published":{"date-parts":[[2019,5]]}}}