{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,9,25]],"date-time":"2025-09-25T14:05:57Z","timestamp":1758809157044,"version":"3.28.0"},"reference-count":22,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,5]]},"DOI":"10.1109\/icassp.2019.8682224","type":"proceedings-article","created":{"date-parts":[[2019,4,17]],"date-time":"2019-04-17T20:01:56Z","timestamp":1555531316000},"page":"7100-7104","source":"Crossref","is-referenced-by-count":8,"title":["Windowed Attention Mechanisms for Speech Recognition"],"prefix":"10.1109","author":[{"given":"Shucong","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Erfan","family":"Loweimi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Peter","family":"Bell","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Steve","family":"Renals","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2017-1296","article-title":"Advances in joint CTC-attention based end-to-end speech recognition with a deep CNN encoder and RNN-LM","author":"hori","year":"2017"},{"key":"ref11","first-page":"5067","article-title":"An online sequence-to-sequence model using partial conditioning","author":"jaitly","year":"2016","journal-title":"NIPS"},{"year":"2017","author":"sainath","article-title":"Improving the performance of online neural transducer models","key":"ref12"},{"doi-asserted-by":"publisher","key":"ref13","DOI":"10.1109\/ICASSP.2016.7472618"},{"year":"2017","author":"tjandra","article-title":"Local monotonic attention mechanism for end-to-end speech and language processing","key":"ref14"},{"key":"ref15","article-title":"DARPA TIMIT acoustic-phonetic continuous speech corpus CD-ROM. NIST speech disc 1-1.1","volume":"93","author":"garofolo","year":"1993","journal-title":"NASA STI\/Recon Technical Report N"},{"year":"1994","article-title":"CSR-II (wsj1) complete","key":"ref16"},{"key":"ref17","article-title":"The Kaldi speech recognition toolkit","author":"povey","year":"2011","journal-title":"IEEE ASRU"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.21437\/Interspeech.2018-1456"},{"year":"2015","author":"chan","article-title":"Listen, attend and spell","key":"ref19"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"939","DOI":"10.21437\/Interspeech.2017-233","article-title":"A comparison of sequence-to-sequence models for speech recognition","author":"prabhavalkar","year":"2017","journal-title":"InterSpeech"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.18653\/v1\/D15-1166"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.21437\/Interspeech.2018-1616"},{"key":"ref5","first-page":"577","article-title":"Attention-based models for speech recognition","author":"chorowski","year":"2015","journal-title":"NIPS"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.21437\/Interspeech.2017-232"},{"key":"ref7","doi-asserted-by":"crossref","first-page":"523","DOI":"10.21437\/Interspeech.2017-343","article-title":"Towards better decoding and language model integration in sequence to sequence models","author":"chorowski","year":"2017","journal-title":"InterSpeech"},{"year":"2014","author":"bahdanau","article-title":"Neural machine translation by jointly learning to align and translate","key":"ref2"},{"key":"ref1","first-page":"2048","article-title":"Show, attend and tell: Neural image caption generation with visual attention","author":"xu","year":"2015","journal-title":"ICML"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.1109\/ICASSP.2017.7953075"},{"year":"2012","author":"zeiler","article-title":"Adadelta: an adaptive learning rate method","key":"ref20"},{"year":"2016","author":"van den oord","article-title":"Wavenet: A generative model for raw audio","key":"ref22"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.1186\/s13636-015-0068-3"}],"event":{"name":"ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2019,5,12]]},"location":"Brighton, United Kingdom","end":{"date-parts":[[2019,5,17]]}},"container-title":["ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8671773\/8682151\/08682224.pdf?arnumber=8682224","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,15]],"date-time":"2022-07-15T03:13:27Z","timestamp":1657854807000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8682224\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5]]},"references-count":22,"URL":"https:\/\/doi.org\/10.1109\/icassp.2019.8682224","relation":{},"subject":[],"published":{"date-parts":[[2019,5]]}}}