{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,25]],"date-time":"2026-03-25T14:24:37Z","timestamp":1774448677612,"version":"3.50.1"},"reference-count":30,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,12,13]],"date-time":"2021-12-13T00:00:00Z","timestamp":1639353600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,12,13]],"date-time":"2021-12-13T00:00:00Z","timestamp":1639353600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62071461,11774380"],"award-info":[{"award-number":["62071461,11774380"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,12,13]]},"DOI":"10.1109\/asru51503.2021.9688090","type":"proceedings-article","created":{"date-parts":[[2022,2,3]],"date-time":"2022-02-03T15:31:00Z","timestamp":1643902260000},"page":"1003-1010","source":"Crossref","is-referenced-by-count":4,"title":["Far-Field Speech Recognition Based on Complex-Valued Neural Networks and Inter-Frame Similarity Difference Method"],"prefix":"10.1109","author":[{"given":"Yifan","family":"Guo","sequence":"first","affiliation":[{"name":"Institute of Acoustics, Chinese Academy of Sciences,Key Laboratory of Speech Acoustics and Content Understanding"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yifan","family":"Chen","sequence":"additional","affiliation":[{"name":"Institute of Acoustics, Chinese Academy of Sciences,Key Laboratory of Speech Acoustics and Content Understanding"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Gaofeng","family":"Cheng","sequence":"additional","affiliation":[{"name":"Institute of Acoustics, Chinese Academy of Sciences,Key Laboratory of Speech Acoustics and Content Understanding"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Pengyuan","family":"Zhang","sequence":"additional","affiliation":[{"name":"Institute of Acoustics, Chinese Academy of Sciences,Key Laboratory of Speech Acoustics and Content Understanding"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yonghong","family":"Yan","sequence":"additional","affiliation":[{"name":"Institute of Acoustics, Chinese Academy of Sciences,Key Laboratory of Speech Acoustics and Content Understanding"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854048"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383492"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/PROC.1969.7278"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2821903"},{"key":"ref13","author":"haykin","year":"2008","journal-title":"Adaptive Filter Theory"},{"key":"ref14","first-page":"6000","article-title":"Attention is all you need","author":"ashish","year":"0","journal-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems"},{"key":"ref15","first-page":"4232","article-title":"Complex trans-former: A framework for modeling complex-valued se-quence","author":"muqiao","year":"0","journal-title":"ICASSP 2020 &#x2014; 2020 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref16","first-page":"836","article-title":"Channel-attention dense u-net for multichannel speech enhance-ment","author":"bahareh","year":"2020","journal-title":"ICASSP 2020&#x2013;2020 IEEE International Con-ference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref17","doi-asserted-by":"crossref","first-page":"1240","DOI":"10.1109\/JSTSP.2017.2763455","article-title":"Hybrid etc\/attention architecture for end-to-end speech recognition","volume":"11","author":"shinji","year":"2017","journal-title":"IEEE Journal of Selected Topics in Signal Processing"},{"key":"ref18","article-title":"How much position information do convolutional neural net-works encode?","author":"islam","year":"0","journal-title":"International Conference on Learning Representations"},{"key":"ref19","first-page":"14274","article-title":"On translation invariance in cnns: Convolutional layers can exploit absolute spatial location","author":"kayhan","year":"0","journal-title":"Proceedings of the IEEE\/CVF Conference on Computer Vision and Pattern Recognition"},{"key":"ref28","article-title":"Musan: A music, speech, and noise corpus","author":"snyder","year":"2015","journal-title":"ArXiv Preprint"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2004.07.028"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2009.2025790"},{"key":"ref3","first-page":"324","article-title":"End-to-end far-field speech recognition with unified derever-beration and beamforming","author":"zhang","year":"0","journal-title":"Proc Interspeech 2020"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2672401"},{"key":"ref29","first-page":"5220","article-title":"A study on data augmentation of reverberant speech for robust speech recognition","author":"tom","year":"2017","journal-title":"2017 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404770"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053940"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682977"},{"key":"ref2","first-page":"2632","article-title":"Multichannel end-to-end speech recognition","author":"ochiai","year":"2017","journal-title":"International Conference on Machine Learning"},{"key":"ref9","first-page":"334","article-title":"Improved guided source separation in-tegrated with a strong back-end for the chime-6 dinner party scenario","author":"chen","year":"0","journal-title":"Proc Interspeech 2020"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-662-04619-7"},{"key":"ref20","author":"fant","year":"1970","journal-title":"Acoustic Theory of Speech Production"},{"key":"ref22","author":"kamo","year":"0","journal-title":"An end-to-end multichan-nel asr based on dnn mvdr beamformer"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1002\/9781118142882"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1121\/1.382599"},{"key":"ref26","article-title":"Neural net-work based spectral mask estimation for acoustic beam-forming","author":"haeb-umbach","year":"0","journal-title":"Proc IEEE Intl Conf on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref25","article-title":"Demand: a collection of multi-channel recordings of acoustic noise in diverse environments","author":"thiemann","year":"0","journal-title":"Proc Meetings Acoust"}],"event":{"name":"2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","location":"Cartagena, Colombia","start":{"date-parts":[[2021,12,13]]},"end":{"date-parts":[[2021,12,17]]}},"container-title":["2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9687821\/9687855\/09688090.pdf?arnumber=9688090","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,13]],"date-time":"2026-02-13T20:48:33Z","timestamp":1771015713000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9688090\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,12,13]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/asru51503.2021.9688090","relation":{},"subject":[],"published":{"date-parts":[[2021,12,13]]}}}