{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,25]],"date-time":"2026-07-25T16:32:04Z","timestamp":1784997124238,"version":"3.55.0"},"reference-count":184,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"2","license":[{"start":{"date-parts":[[2021,2,1]],"date-time":"2021-02-01T00:00:00Z","timestamp":1612137600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"}],"funder":[{"DOI":"10.13039\/100006785","name":"Google Faculty Research Award","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100006785","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["Proc. IEEE"],"published-print":{"date-parts":[[2021,2]]},"DOI":"10.1109\/jproc.2020.3018668","type":"journal-article","created":{"date-parts":[[2020,9,9]],"date-time":"2020-09-09T20:11:46Z","timestamp":1599682306000},"page":"124-148","source":"Crossref","is-referenced-by-count":92,"title":["Far-Field Automatic Speech Recognition"],"prefix":"10.1109","volume":"109","author":[{"ORCID":"https:\/\/orcid.org\/0000-0001-9468-7330","authenticated-orcid":false,"given":"Reinhold","family":"Haeb-Umbach","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jahn","family":"Heymann","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-3683-5432","authenticated-orcid":false,"given":"Lukas","family":"Drude","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5970-8631","authenticated-orcid":false,"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5175-7834","authenticated-orcid":false,"given":"Marc","family":"Delcroix","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Tomohiro","family":"Nakatani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref170","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2655259"},{"key":"ref172","first-page":"1","article-title":"MARVELO - A Framework for Signal Processing in Wireless Acoustic Sensor Networks","author":"afifi","year":"2018","journal-title":"IEEE Speech Communication"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462458"},{"key":"ref174","first-page":"1","article-title":"Distributed MVDR beamforming for (wireless) microphone networks using message passing","author":"heusdens","year":"2012","journal-title":"Proc Intl Workshop Acoust Signal Enhancement (IWAENC)"},{"key":"ref173","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2011.2169409"},{"key":"ref176","doi-asserted-by":"publisher","DOI":"10.1109\/SCVT.2011.6101302"},{"key":"ref175","doi-asserted-by":"publisher","DOI":"10.1016\/j.sigpro.2014.07.014"},{"key":"ref178","doi-asserted-by":"publisher","DOI":"10.1007\/s10489-014-0629-7"},{"key":"ref177","doi-asserted-by":"publisher","DOI":"10.1109\/HSCMA.2011.5942398"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.1109\/MMSE.2004.79"},{"key":"ref169","doi-asserted-by":"publisher","DOI":"10.1109\/ASPAA.2009.5346505"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/53.665"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1994.389749"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2010.5495701"},{"key":"ref32","year":"2008","journal-title":"ITU-T Recommendation P 862 Perceptual Evaluation of Speech Quality (PESQ) An Objective Method for End-to-end Speech Quality Assessment of Narrow-band Telephone Networks and Speech Codecs"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2892412"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2009.2020891"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.23919\/EUSIPCO.2019.8902753"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2004.1326688"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2842159"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2005.858005"},{"key":"ref181","article-title":"The Kaldi speech recognition toolkit","author":"povey","year":"2011","journal-title":"Proc IEEE Workshop Autom Speech Recognition Understanding (ASRU)"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461310"},{"key":"ref184","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.902460"},{"key":"ref183","first-page":"28","article-title":"The AMI meeting corpus: A pre-announcement","author":"carletta","year":"2005","journal-title":"Proc 2nd Int Workshop Mach Learn Multimodal Interact"},{"key":"ref182","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1456"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2006.888292"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2915167"},{"key":"ref179","article-title":"NARA-WPE: A Python package for weighted prediction error dereverberation in numpy and Tensorflow for online and offline processing","author":"drude","year":"2018","journal-title":"Proc ITG"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/78.149989"},{"key":"ref20","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-84996-056-4","author":"naylor","year":"2010","journal-title":"Speech Dereverberation"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471702"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2052251"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471631"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2623559"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054345"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2726762"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003750"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952155"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1109\/TAP.1982.1142739"},{"key":"ref51","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4020-6479-1","volume":"615","author":"makino","year":"2007","journal-title":"Blind Speech Separation"},{"key":"ref154","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2064307"},{"key":"ref153","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2164527"},{"key":"ref156","first-page":"509","article-title":"The ICSI RT07s speaker diarization system","author":"wooters","year":"2007","journal-title":"Multimodal Technologies for Perception of Humans"},{"key":"ref155","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref150","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639096"},{"key":"ref152","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2125954"},{"key":"ref151","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6637694"},{"key":"ref146","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462105"},{"key":"ref147","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1616"},{"key":"ref148","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"ref149","doi-asserted-by":"publisher","DOI":"10.1109\/97.736233"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953173"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/IWAENC.2018.8521255"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683294"},{"key":"ref56","first-page":"384","article-title":"Neural network-based spectrum estimation for online WPE dereverberation","author":"kinoshita","year":"2017","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref55","article-title":"Tight integration of spatial and spectral features for BSS with deep clustering embeddings","author":"drude","year":"2017","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472778"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471664"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-55016-4"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2647702"},{"key":"ref167","first-page":"2629","article-title":"The DIRHA simulated corpus","author":"cristoforetti","year":"2014","journal-title":"Proc LREC"},{"key":"ref166","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053921"},{"key":"ref165","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053845"},{"key":"ref164","article-title":"Santa Barbara corpus of spoken American English","author":"du bois","year":"0","journal-title":"LDC94S21 CD-ROM Philadelphia Linguistic Data Consortium"},{"key":"ref163","article-title":"The LOCATA challenge: Acoustic source localization and tracking","author":"evers","year":"2019","journal-title":"arXiv 1909 01008"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.1163\/016918610X493561"},{"key":"ref161","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461669"},{"key":"ref160","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2665341"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1768"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2016.11.005"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404843"},{"key":"ref5","article-title":"CHiME-6 challenge: Tackling multispeaker speech recognition for unsegmented recordings","author":"watanabe","year":"2020","journal-title":"arXiv 2004 09249"},{"key":"ref159","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003959"},{"key":"ref8","article-title":"The CMU-MIT REVERB challenge 2014 system: Description and results","author":"feng","year":"2014","journal-title":"Proc REVERB Challenge Workshop"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1176"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1186\/s13634-015-0245-7"},{"key":"ref157","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2014.7078610"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682572"},{"key":"ref9","article-title":"The MERL\/MELCO\/TUM system for the REVERB challenge using deep recurrent neural network feature enhancement","author":"weninger","year":"2014","journal-title":"Proc REVERB Challenge Workshop"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2012.2210879"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1186\/s13634-015-0256-4"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472671"},{"key":"ref47","first-page":"3877","article-title":"Adaptive multichannel dereverberation for automatic speech recognition","author":"caroselli","year":"2017","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/78.558487"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054393"},{"key":"ref44","doi-asserted-by":"crossref","DOI":"10.1007\/978-3-319-73031-8","author":"makino","year":"2018","journal-title":"Audio Source Separation"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1002\/9781119279860"},{"key":"ref127","article-title":"When mismatched training data outperform matched data","author":"vincent","year":"2017","journal-title":"Systematic approaches to deep learning methods for audio"},{"key":"ref126","first-page":"2992","article-title":"Is speech enhancement pre-processing still relevant when using deep neural networks for acoustic modeling?","author":"delcroix","year":"2013","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1475"},{"key":"ref124","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682273"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178061"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2008.09.001"},{"key":"ref129","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639212"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404828"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.1162\/NECO_a_00697"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/EUSIPCO.2016.7760429"},{"key":"ref76","first-page":"2655","article-title":"Speaker-aware neural network based beamformer for speaker extraction in speech mixtures","author":"\u017emol\u00edkov\u00e1","year":"2017","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639201"},{"key":"ref77","doi-asserted-by":"crossref","first-page":"800","DOI":"10.1109\/JSTSP.2019.2922820","article-title":"SpeakerBeam: Speaker aware neural network for target speaker extraction in speech mixtures","volume":"13","author":"\u017emol\u00edkov\u00e1","year":"2019","journal-title":"IEEE J Sel Topics Signal Process"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404829"},{"key":"ref75","first-page":"1981","article-title":"Improved MVDR beamforming using single-channel mask prediction networks","author":"erdogan","year":"2016","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462372"},{"key":"ref134","first-page":"577","article-title":"Attention-based models for speech recognition","author":"chorowski","year":"2015","journal-title":"Proc Adv Neural Inf Process Syst (NeurIPS)"},{"key":"ref131","first-page":"1","article-title":"Multi-channel speech recognition: LSTMs all the way through","author":"erdogan","year":"2016","journal-title":"Proc CHiME Workshop"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472692"},{"key":"ref132","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952756"},{"key":"ref79","article-title":"VoiceFilter: Targeted voice separation by speaker-conditioned spectrogram masking","author":"wang","year":"2018","journal-title":"arXiv 1810 04826"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"ref135","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref138","first-page":"1764","article-title":"Multichannel end-to-end speech recognition","author":"ochiai","year":"2017","journal-title":"Proc Int Conf Mach Learn (ICML)"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2438549"},{"key":"ref139","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2017.2764276"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/29.1509"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2906427"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2804172"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/78.934132"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1301"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2372335"},{"key":"ref141","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682650"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2008.4517549"},{"key":"ref142","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2019.8937250"},{"key":"ref67","first-page":"953","article-title":"An EM algorithm for localizing multiple sound sources in reverberant environments","author":"mandel","year":"2007","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref143","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003986"},{"key":"ref68","first-page":"241","article-title":"Blind speech separation employing directional statistics in an expectation maximization framework","author":"vu","year":"2010","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process (ICASSP)"},{"key":"ref144","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1856"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2016.10.005"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471718"},{"key":"ref145","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1130"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1186\/s13634-016-0306-6"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.1109\/ICDSP.2009.5201259"},{"key":"ref95","first-page":"399","article-title":"Acoustic modeling for Google home","author":"li","year":"2017","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref108","article-title":"Acoustical sound database in real environments for sound scene understanding and hands-free speech recognition","author":"nakamura","year":"2000","journal-title":"Proc Int Conf Lang Res Eval (LREC)"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref107","doi-asserted-by":"publisher","DOI":"10.21437\/CHiME.2018-2"},{"key":"ref93","author":"huang","year":"2001","journal-title":"Spoken Language Processing A Guide to Theory Algorithm and System Development"},{"key":"ref106","first-page":"2751","article-title":"Purely sequence-trained neural networks for ASR based on lattice-free MMI","author":"povey","year":"2016","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/5.18626"},{"key":"ref105","first-page":"2345","article-title":"Sequence-discriminative training of deep neural networks","author":"vesel?","year":"2013","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-540-49127-9_26"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.21437\/CHiME.2018-3"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2284"},{"key":"ref103","article-title":"The USTC-iFlytek System for CHiME-4 Challenge","author":"du","year":"2016","journal-title":"Proc CHiME Workshop"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2010-343"},{"key":"ref111","doi-asserted-by":"publisher","DOI":"10.1121\/1.382599"},{"key":"ref112","article-title":"Room impulse response generator","author":"habets","year":"2006"},{"key":"ref110","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2019.2917582"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178838"},{"key":"ref99","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1109\/29.21701"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178920"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2975902"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2017.02.007"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1262"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003785"},{"key":"ref14","article-title":"The USTC-NELSLIP systems for CHiME-6 challenge","author":"du","year":"2020","journal-title":"Proc 6th CHiME Speech Separat Recognit Challenge (CHiME)"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref118","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2019.08.006"},{"key":"ref16","doi-asserted-by":"crossref","DOI":"10.1007\/978-1-4471-5779-3","author":"yu","year":"2015","journal-title":"Automatic Speech Recognition?A Deep Learning Approach"},{"key":"ref82","article-title":"Permutation invariant training of deep models for speaker-independent multi-talker speech separation","author":"yu","year":"2016","journal-title":"arXiv 1607 00325 [cs]"},{"key":"ref117","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462624"},{"key":"ref17","author":"li","year":"2015","journal-title":"Robustness in Automatic Speech Recogniton"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2004.832994"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2009.2025790"},{"key":"ref84","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683198"},{"key":"ref119","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1016\/0167-6393(90)90019-6"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683520"},{"key":"ref114","first-page":"3586","article-title":"Audio augmentation for speech recognition","author":"ko","year":"2015","journal-title":"Proc Annu Conf Int Speech Commun Assoc (Interspeech)"},{"key":"ref113","article-title":"MUSAN: A music, speech, and noise corpus","author":"snyder","year":"2015","journal-title":"arXiv 1510 08484 [cs]"},{"key":"ref116","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1247"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.21437\/CHiME.2018-8"},{"key":"ref115","first-page":"16","article-title":"Unsupervised domain adaptation for robust speech recognition via variational autoencoder-based data augmentation","author":"hsu","year":"2017","journal-title":"Proc IEEE Workshop Autom Speech Recognition Understanding (ASRU)"},{"key":"ref120","article-title":"The STC system for the CHiME-6 challenge","author":"medennikov","year":"2020","journal-title":"Proc 6th CHiME Speech Separat Recognit Challenge (CHiME)"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2196"},{"key":"ref121","doi-asserted-by":"publisher","DOI":"10.1109\/ICDSP.2018.8631862"},{"key":"ref122","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2030"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682556"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683201"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461639"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952163"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2019.2912565"}],"container-title":["Proceedings of the IEEE"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/5\/9328628\/09189820.pdf?arnumber=9189820","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,10,7]],"date-time":"2023-10-07T02:58:00Z","timestamp":1696647480000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9189820\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,2]]},"references-count":184,"journal-issue":{"issue":"2"},"URL":"https:\/\/doi.org\/10.1109\/jproc.2020.3018668","relation":{},"ISSN":["0018-9219","1558-2256"],"issn-type":[{"value":"0018-9219","type":"print"},{"value":"1558-2256","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,2]]}}}