{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,6]],"date-time":"2026-06-06T16:46:17Z","timestamp":1780764377567,"version":"3.54.1"},"reference-count":44,"publisher":"IEEE","funder":[{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,1,19]]},"DOI":"10.1109\/slt48900.2021.9383583","type":"proceedings-article","created":{"date-parts":[[2021,3,25]],"date-time":"2021-03-25T20:46:54Z","timestamp":1616705214000},"page":"30-37","source":"Crossref","is-referenced-by-count":3,"title":["Convolution-Based Attention Model With Positional Encoding For Streaming Speech Recognition On Embedded Devices"],"prefix":"10.1109","author":[{"given":"Jinhwan","family":"Park","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chanwoo","family":"Kim","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Wonyong","family":"Sung","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref39","first-page":"3935","article-title":"Enhancing the TED-LIUM corpus with selected data for language modeling and more ted talks","author":"rousseau","year":"2014","journal-title":"LREC"},{"key":"ref38","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2020-1069","article-title":"CTC-synchronous training for monotonic attention model","author":"inaguma","year":"2020"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref32","first-page":"779","article-title":"You only look once: Unified, real-time object detection","author":"redmon","year":"2016","journal-title":"Proceedings of the IEEE Conference on Computer Vision and Pattern Recognition"},{"key":"ref31","article-title":"Very deep convolutional networks for large-scale image recognition","author":"simonyan","year":"2014"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2460"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-751"},{"key":"ref36","first-page":"431","article-title":"Local monotonic attention mechanism for end-to-end speech and language processing","author":"tjandra","year":"2017","journal-title":"Proceedings of the Eighth International Joint Conference on Natural Language Processing (Volume 1 Long Papers)"},{"key":"ref35","article-title":"Layer normalization","author":"ba","year":"2016"},{"key":"ref34","article-title":"How much position information do convolutional neural networks encode?","author":"islam","year":"2020","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D18-1477"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P16-1162"},{"key":"ref11","article-title":"Parallelizing linear recurrent neural nets over sequence length","author":"martin","year":"2018","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref12","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2702"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054476"},{"key":"ref15","doi-asserted-by":"crossref","first-page":"523","DOI":"10.21437\/Interspeech.2017-343","article-title":"Towards better decoding and language model integration in sequence to sequence models","author":"chorowski","year":"2017","journal-title":"Proc Interspeech 2017"},{"key":"ref16","first-page":"577","article-title":"Attention-based models for speech recognition","author":"chorowski","year":"2015","journal-title":"Advances in neural information processing systems"},{"key":"ref17","article-title":"Monotonic chunkwise attention","author":"chiu","year":"2018","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682336"},{"key":"ref28","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2020-2840","article-title":"Scaling up online speech recognition using convnets","author":"pratap","year":"2020"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref27","article-title":"Transformer-based acoustic modeling for hybrid speech recognition","author":"wang","year":"2019"},{"key":"ref3","first-page":"173","article-title":"Deep speech 2: End-to-end speech recognition in english and mandarin","author":"amodei","year":"2016","journal-title":"International Conference on Machine Learning"},{"key":"ref6","first-page":"206","article-title":"Exploring neural transducers for end-to-end speech recognition","author":"battenberg","year":"0","journal-title":"Automatic Speech Recognition and Understanding (ASRU) 2017 IEEE Workshop on"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053054"},{"key":"ref5","doi-asserted-by":"crossref","first-page":"193","DOI":"10.1109\/ASRU.2017.8268935","article-title":"Exploring architectures, data and units for streaming end-to-end speech recognition with RNN-transducer","author":"rao","year":"2017","journal-title":"IEEE Workshop on Automatic Speech Recognition and Understanding (ASRU)"},{"key":"ref8","first-page":"1243","article-title":"Convolutional sequence to sequence learning","author":"gehring","year":"2017","journal-title":"International Conference on Machine Learning (ICML)"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1145\/3079856.3080246"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"ref9","article-title":"Quasi-recurrent neural networks","author":"bradbury","year":"2017","journal-title":"International Conference on Learning Representations (ICLR)"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462105"},{"key":"ref20","first-page":"939","article-title":"A comparison of sequence-to-sequence models for speech recognition","author":"prabhavalkar","year":"0"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9004027"},{"key":"ref21","first-page":"2837","article-title":"Online and linear-time attention by enforcing monotonic alignments","volume":"70","author":"raffel","year":"2017","journal-title":"ICML&#x2019;17 Proceedings of the 34th International Conference on Machine Learning"},{"key":"ref42","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/W14-4012"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9004025"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953176"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref25","article-title":"Letter-based speech recognition with gated ConvNets","author":"liptchinsky","year":"2017"}],"event":{"name":"2021 IEEE Spoken Language Technology Workshop (SLT)","location":"Shenzhen, China","start":{"date-parts":[[2021,1,19]]},"end":{"date-parts":[[2021,1,22]]}},"container-title":["2021 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9383468\/9383452\/09383583.pdf?arnumber=9383583","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,22]],"date-time":"2022-12-22T13:16:18Z","timestamp":1671714978000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9383583\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,1,19]]},"references-count":44,"URL":"https:\/\/doi.org\/10.1109\/slt48900.2021.9383583","relation":{},"subject":[],"published":{"date-parts":[[2021,1,19]]}}}