{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,23]],"date-time":"2026-04-23T11:08:54Z","timestamp":1776942534708,"version":"3.51.4"},"reference-count":81,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,1,1]],"date-time":"2022-01-01T00:00:00Z","timestamp":1640995200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"National Key Research and Development Program of China","award":["2021ZD0201504"],"award-info":[{"award-number":["2021ZD0201504"]}]},{"name":"China NSFC","award":["62122050"],"award-info":[{"award-number":["62122050"]}]},{"name":"China NSFC","award":["62071288"],"award-info":[{"award-number":["62071288"]}]},{"name":"Shanghai Municipal Science and Technology Major Project","award":["2021SHZDZX0102"],"award-info":[{"award-number":["2021SHZDZX0102"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2022]]},"DOI":"10.1109\/taslp.2022.3209942","type":"journal-article","created":{"date-parts":[[2022,9,27]],"date-time":"2022-09-27T19:52:37Z","timestamp":1664308357000},"page":"3173-3188","source":"Crossref","is-referenced-by-count":22,"title":["End-to-End Dereverberation, Beamforming, and Speech Recognition in a Cocktail Party"],"prefix":"10.1109","volume":"30","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4500-3515","authenticated-orcid":false,"given":"Wangyou","family":"Zhang","sequence":"first","affiliation":[{"name":"X-LANCE Lab, Department of Computer Science and Engineering and MoE Key Laboratory of Artificial Intelligence, AI Institute, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5221-5412","authenticated-orcid":false,"given":"Xuankai","family":"Chang","sequence":"additional","affiliation":[{"name":"Language Technologies Institute, Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Christoph","family":"Boeddeker","sequence":"additional","affiliation":[{"name":"Paderborn University, Paderborn, Germany"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-7487-7150","authenticated-orcid":false,"given":"Tomohiro","family":"Nakatani","sequence":"additional","affiliation":[{"name":"NTT Corporation, Kyoto, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5970-8631","authenticated-orcid":false,"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[{"name":"Language Technologies Institute, Carnegie Mellon University, Pittsburgh, PA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-0314-3790","authenticated-orcid":false,"given":"Yanmin","family":"Qian","sequence":"additional","affiliation":[{"name":"X-LANCE Lab, Department of Computer Science and Engineering and MoE Key Laboratory of Artificial Intelligence, AI Institute, Shanghai Jiao Tong University, Shanghai, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461870"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1114"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2018.09.003"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/CHiME.2020-1"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1121\/1.1907229"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-305"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461893"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053461"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639593"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053092"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471664"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2016-552"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/53.665"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413594"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682576"},{"key":"ref16","first-page":"2632","article-title":"Multichannel end-to-end speech recognition","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ochiai","year":"2017"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2017.8282233"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953173"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2019.8937250"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2172"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003986"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054029"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461639"},{"key":"ref24","article-title":"SMS-WSJ: Database, performance measures, and baseline recipe for multi-channel source separation and recognition","author":"Drude","year":"2019"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053327"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2432"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414464"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2008.4517552"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2006.888292"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2052251"},{"key":"ref31","article-title":"Leveraging low-distortion target estimates for improved speech enhancement","author":"Wang","year":"2021"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462430"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA52581.2021.9632720"},{"key":"ref34","first-page":"26","article-title":"A study of learning based beamforming methods for speech recognition","volume-title":"Proc. CHiME Workshop","author":"Xiao","year":"2016"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683294"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/JPROC.2020.3018668"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/IROS47612.2022.9981659"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/PROC.1969.7278"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1002\/0471221082"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2019.2911179"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054393"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1458"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383528"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383522"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-733"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2196"},{"key":"ref47","first-page":"1","article-title":"NARA-WPE: A python package for weighted prediction error dereverberation in Numpy and Tensorflow for online and offline processing","volume-title":"Proc. 13th ITG- Symp.","author":"Drude","year":"2018"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2009.2016395"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1002\/zamm.19290090206"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2047324"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471631"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952154"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2726762"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2015.2513004"},{"key":"ref56","article-title":"An investigation of end-to-end multichannel speech recognition for reverberant and mismatch conditions","author":"Subramanian"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1002\/j.1538-7305.1955.tb01488.x"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1049\/ip-rsn:20041069"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2008.2005862"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/JOE.2018.2815480"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/TAC.1974.1100466"},{"key":"ref62","article-title":"The matrix cookbook","author":"Petersen","year":"2012"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1002\/nla.1890"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-721"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1739"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2002.5743880"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2004.832994"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2018.11.005"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-662-04619-7_8"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2876169"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682822"},{"key":"ref72","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kingma","year":"2015"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639693"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2005.858005"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2114881"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683855"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414661"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-220"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003750"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-2253"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9657755\/09904314.pdf?arnumber=9904314","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,1,22]],"date-time":"2024-01-22T22:13:25Z","timestamp":1705961605000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9904314\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022]]},"references-count":81,"URL":"https:\/\/doi.org\/10.1109\/taslp.2022.3209942","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022]]}}}