{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,28]],"date-time":"2025-10-28T10:54:47Z","timestamp":1761648887193,"version":"3.28.0"},"reference-count":26,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,6,6]],"date-time":"2021-06-06T00:00:00Z","timestamp":1622937600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,6,6]],"date-time":"2021-06-06T00:00:00Z","timestamp":1622937600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,6,6]]},"DOI":"10.1109\/icassp39728.2021.9413493","type":"proceedings-article","created":{"date-parts":[[2021,5,13]],"date-time":"2021-05-13T19:53:45Z","timestamp":1620935625000},"page":"6214-6218","source":"Crossref","is-referenced-by-count":3,"title":["Gaussian Kernelized Self-Attention for Long Sequence Data and its Application to CTC-Based Speech Recognition"],"prefix":"10.1109","author":[{"given":"Yosuke","family":"Kashiwagi","sequence":"first","affiliation":[{"name":"Sony Corporation,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Emiru","family":"Tsunoo","sequence":"additional","affiliation":[{"name":"Sony Corporation,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[{"name":"Johns Hopkins University,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"doi-asserted-by":"publisher","key":"ref10","DOI":"10.1109\/ICASSP.2019.8682539"},{"doi-asserted-by":"publisher","key":"ref11","DOI":"10.21437\/Interspeech.2018-1910"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.18653\/v1\/N18-2074"},{"year":"2020","author":"pham","article-title":"Relative positional en-coding for speech recognition and direct translation","key":"ref13"},{"year":"2020","author":"kitaev","article-title":"Reformer: The Efficient Transformer","key":"ref14"},{"doi-asserted-by":"publisher","key":"ref15","DOI":"10.1109\/CVPR.2016.41"},{"key":"ref16","first-page":"1509","article-title":"Locality-sensitive binary codes from shift-invariant kernels","author":"raginsky","year":"2009","journal-title":"Advances in neural information processing systems"},{"key":"ref17","doi-asserted-by":"crossref","first-page":"317","DOI":"10.1109\/JSTARS.2013.2262926","article-title":"A kernel-based feature selection method for SVM with RBF kernel for hyperspectral image classification","volume":"7","author":"kuo","year":"2013","journal-title":"IEEE Journal of Selected Topics in Applied Earth Observations and Remote Sensing"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.1109\/ICACDOT.2016.7877753"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1109\/ISCAS.2005.1465466"},{"year":"2019","author":"mohamed","article-title":"Transformers with convolutional context for ASR","key":"ref4"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1109\/ASRU46091.2019.9003750"},{"doi-asserted-by":"publisher","key":"ref6","DOI":"10.1109\/ICASSP40776.2020.9054029"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.1109\/ASRU46091.2019.9004025"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1109\/ICASSP40776.2020.9054345"},{"doi-asserted-by":"publisher","key":"ref7","DOI":"10.1109\/ICASSP.2018.8462497"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1109\/ICASSP.2018.8462506"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.21437\/Interspeech.2019-2702"},{"key":"ref1","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Advances in neural information processing systems"},{"key":"ref20","article-title":"A hybrid SVM\/HMM acoustic modeling approach to automatic speech recognition","author":"stadermann","year":"2004","journal-title":"Proc Int Conf on Spoken Language Processing ICSLP# 2004"},{"key":"ref22","article-title":"Corpus of spontaneous japanese: Its design and evaluation","author":"maekawa","year":"2003","journal-title":"Proceedings of SSPR"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.1109\/TSP.2004.831018"},{"doi-asserted-by":"publisher","key":"ref24","DOI":"10.1145\/1143844.1143891"},{"key":"ref23","doi-asserted-by":"crossref","first-page":"198","DOI":"10.1007\/978-3-319-99579-3_21","article-title":"Ted-lium 3: twice as much data and corpus repartition for experiments on speaker adaptation","author":"hernandez","year":"2018","journal-title":"International Conference on Speech and Computer"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.18653\/v1\/P18-1007"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.21437\/Interspeech.2019-2680"}],"event":{"name":"ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2021,6,6]]},"location":"Toronto, ON, Canada","end":{"date-parts":[[2021,6,11]]}},"container-title":["ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9413349\/9413350\/09413493.pdf?arnumber=9413493","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,12,27]],"date-time":"2022-12-27T08:30:45Z","timestamp":1672129845000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9413493\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,6]]},"references-count":26,"URL":"https:\/\/doi.org\/10.1109\/icassp39728.2021.9413493","relation":{},"subject":[],"published":{"date-parts":[[2021,6,6]]}}}