{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T15:47:55Z","timestamp":1784389675676,"version":"3.55.0"},"reference-count":296,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,1,1]],"date-time":"2021-01-01T00:00:00Z","timestamp":1609459200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100021851","name":"William Demant Fonden","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100021851","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2021]]},"DOI":"10.1109\/taslp.2021.3066303","type":"journal-article","created":{"date-parts":[[2021,3,17]],"date-time":"2021-03-17T20:07:48Z","timestamp":1616011668000},"page":"1368-1396","source":"Crossref","is-referenced-by-count":285,"title":["An Overview of Deep-Learning-Based Audio-Visual Speech Enhancement and Separation"],"prefix":"10.1109","volume":"29","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3575-1600","authenticated-orcid":false,"given":"Daniel","family":"Michelsanti","sequence":"first","affiliation":[{"name":"Department of Electronic Systems, Aalborg University, Aalborg, Denmark"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6856-8928","authenticated-orcid":false,"given":"Zheng-Hua","family":"Tan","sequence":"additional","affiliation":[{"name":"Department of Electronic Systems, Aalborg University, Aalborg, Denmark"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shi-Xiong","family":"Zhang","sequence":"additional","affiliation":[{"name":"Tencent AI Laboratory, Bellevue, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yong","family":"Xu","sequence":"additional","affiliation":[{"name":"Tencent AI Laboratory, Bellevue, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Meng","family":"Yu","sequence":"additional","affiliation":[{"name":"Tencent AI Laboratory, Bellevue, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dong","family":"Yu","sequence":"additional","affiliation":[{"name":"Tencent AI Laboratory, Bellevue, WA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Jesper","family":"Jensen","sequence":"additional","affiliation":[{"name":"Department of Electronic Systems, Aalborg University, Aalborg, Denmark"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref275","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2015.2512042"},{"key":"ref274","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2821"},{"key":"ref277","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003983"},{"key":"ref276","doi-asserted-by":"publisher","DOI":"10.1097\/00003446-200506000-00004"},{"key":"ref271","first-page":"596","author":"ward","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref270","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461639"},{"key":"ref273","doi-asserted-by":"publisher","DOI":"10.1109\/5.58337"},{"key":"ref170","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2795749"},{"key":"ref272","doi-asserted-by":"publisher","DOI":"10.1109\/GlobalSIP.2014.7032183"},{"key":"ref172","doi-asserted-by":"publisher","DOI":"10.1109\/ICICSP48821.2019.8958547"},{"key":"ref171","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2915167"},{"key":"ref174","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.906197"},{"key":"ref173","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1166"},{"key":"ref176","author":"massaro","year":"2014","journal-title":"Speech Perception by Ear and Eye A Paradigm for Psychological Inquiry"},{"key":"ref175","doi-asserted-by":"publisher","DOI":"10.3758\/s13423-015-0817-4"},{"key":"ref178","first-page":"2008","article-title":"Conditional generative adversarial networks for speech enhancement and noise-robust speaker verification","author":"michelsanti","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref177","doi-asserted-by":"crossref","first-page":"746","DOI":"10.1038\/264746a0","article-title":"Hearing lips and seeing voices","volume":"264","author":"mcgurk","year":"1976","journal-title":"Nature"},{"key":"ref168","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2928140"},{"key":"ref169","first-page":"674","article-title":"An iterative image registration technique with an application to stereo vision","author":"lucas","year":"0","journal-title":"Proc 7th Int Joint Conf Artif Intell"},{"key":"ref39","first-page":"87","article-title":"Lip reading in the wild","author":"chung","year":"0","journal-title":"Proc Asian Conf Comput Vis"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.367"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1121\/1.1907229"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639593"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952155"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2005.860851"},{"key":"ref267","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2014.2352935"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"ref36","first-page":"1131","article-title":"Lite audio-visual speech enhancement","author":"chuang","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref268","doi-asserted-by":"crossref","first-page":"4006","DOI":"10.21437\/Interspeech.2017-1452","article-title":"Tacotron: Towards end-to-end speech synthesis","author":"wang","year":"2017","journal-title":"Proc INTERSPEECH"},{"key":"ref35","first-page":"577","article-title":"Attention-based models for speech recognition","author":"chorowski","year":"0","journal-title":"Proc 28th Int Conf Neural Inf Process Syst"},{"key":"ref269","article-title":"Deep learning based array processing for speech separation, localization, and recognition","author":"wang","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/D14-1179"},{"key":"ref288","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2831456"},{"key":"ref287","doi-asserted-by":"publisher","DOI":"10.1097\/AUD.0b013e3181d4f251"},{"key":"ref286","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952154"},{"key":"ref285","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v34i05.6489"},{"key":"ref284","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00048-X"},{"key":"ref181","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2019.10.006"},{"key":"ref283","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1458"},{"key":"ref180","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1026"},{"key":"ref282","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00097"},{"key":"ref281","first-page":"2048","article-title":"Show, attend and tell: Neural image caption generation with visual attention","author":"xu","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref280","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683120"},{"key":"ref185","doi-asserted-by":"publisher","DOI":"10.1587\/transinf.2015EDP7457"},{"key":"ref184","doi-asserted-by":"publisher","DOI":"10.1250\/ast.39.263"},{"key":"ref183","doi-asserted-by":"publisher","DOI":"10.1121\/1.1907526"},{"key":"ref182","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682713"},{"key":"ref189","first-page":"2616","article-title":"VoxCeleb: A large-scale speaker identification dataset","author":"nagrani","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref188","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053521"},{"key":"ref187","article-title":"Audio-visual speech inpainting with deep learning","author":"morrone","year":"0","journal-title":"Proc Int Conf Acoust Speech Signal Process"},{"key":"ref186","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682061"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1023\/A:1007379606734"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/PROC.1969.7278"},{"key":"ref179","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682790"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2010.2050650"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2007.383344"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/72.279181"},{"key":"ref21","first-page":"385","article-title":"Perceptual objective listening quality assessment (POLQA), the 3rd generation ITU-T standard for end-to-end speech quality measurement part II - Perceptual model","volume":"61","author":"beerends","year":"2013","journal-title":"J Audio Eng Soc"},{"key":"ref24","first-page":"117","article-title":"The cocktail party phenomenon: A review of research on speech intelligibility in multiple-talker conditions","volume":"86","author":"bronkhorst","year":"2000","journal-title":"Acustica United with Acta Acustica"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.331"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/EUSIPCO.2016.7760550"},{"key":"ref278","article-title":"Multi-modal hybrid deep neural network for speech enhancement","author":"wu","year":"2016"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1163\/000579511X605759"},{"key":"ref279","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2012.6288911"},{"key":"ref293","article-title":"Deep audio-visual learning: A survey","author":"zhu","year":"2020"},{"key":"ref50","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.08.002"},{"key":"ref292","first-page":"570","article-title":"The sound of pixels","author":"zhao","year":"0","journal-title":"Proc Eur Conf Comput Vis"},{"key":"ref51","article-title":"Interleaved multitask learning for audio source separation with independent databases","author":"doire","year":"2019"},{"key":"ref295","article-title":"Visually guided sound source separation using cascaded opponent filter network","author":"zhu","year":"2020","journal-title":"Proc Asian Conf Comput Vis"},{"key":"ref294","article-title":"Separating sounds from a single image","author":"zhu","year":"2020"},{"key":"ref296","volume":"22","author":"zwicker","year":"2013","journal-title":"Psychoacoustics Facts and Models"},{"key":"ref154","first-page":"143","article-title":"Generalization and network design strategies","volume":"19","author":"lecun","year":"1989","journal-title":"Connectionism in Perspective"},{"key":"ref153","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683855"},{"key":"ref156","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054180"},{"key":"ref155","first-page":"1","article-title":"A variance modeling framework based on variational autoencoders for speech enhancement","author":"leglaive","year":"0","journal-title":"Proc IEEE 28th Int Workshop Mach Learn Signal Process"},{"key":"ref150","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2012.192"},{"key":"ref291","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00182"},{"key":"ref152","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2716178"},{"key":"ref290","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2009.2030637"},{"key":"ref151","first-page":"3355","article-title":"Reconstructing intelligible audio speech from visual speech features","author":"le cornu","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref146","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2007.366941"},{"key":"ref147","doi-asserted-by":"publisher","DOI":"10.1145\/3240508.3241911"},{"key":"ref148","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v33i01.33012588"},{"key":"ref149","doi-asserted-by":"publisher","DOI":"10.1109\/ISM.2018.00-19"},{"key":"ref289","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846261"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-64680-0_7"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178061"},{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953127"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1109\/ICCVW.2017.61"},{"key":"ref55","doi-asserted-by":"crossref","first-page":"112:1?112:11","DOI":"10.1145\/3197517.3201357","article-title":"Looking to listen at the cocktail party: A speaker-independent audio-visual model for speech separation","volume":"37","author":"ephrat","year":"2018","journal-title":"ACM Trans Graph"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1984.1164453"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1288\/00005537-194809000-00002"},{"key":"ref52","first-page":"11","article-title":"Feature-wise transformations","volume":"3","author":"dumoulin","year":"2018","journal-title":"&#x201D; Distill"},{"key":"ref40","first-page":"251","article-title":"Out of time: Automated lip sync in the wild","author":"chung","year":"0","journal-title":"Proc Asian Conf Comput Vis"},{"key":"ref167","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2018.2853566"},{"key":"ref166","first-page":"101","article-title":"Le signe de l&#x2019;elevation de la voix","volume":"37","author":"lombard","year":"1911","journal-title":"Ann Mal de L&#x2019;Oreille et du Larynx"},{"key":"ref165","doi-asserted-by":"publisher","DOI":"10.1201\/b14529"},{"key":"ref164","doi-asserted-by":"publisher","DOI":"10.1371\/journal.pone.0196391"},{"key":"ref163","first-page":"21","article-title":"SSD: Single shot multibox detector","author":"liu","year":"0","journal-title":"Proc Eur Conf Comput Vis"},{"key":"ref162","doi-asserted-by":"publisher","DOI":"10.1109\/TSP.2013.2277834"},{"key":"ref161","article-title":"Learn to combine modalities in multimodal deep learning","author":"liu","year":"2018","journal-title":"Proc KDD BigMine"},{"key":"ref160","article-title":"A structured self-attentive sentence embedding","author":"lin","year":"2017","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref4","first-page":"61","article-title":"Towards next-generation lip-reading driven hearing-aids: A preliminary prototype demo","author":"adeel","year":"0","journal-title":"Proc Int Workshop Challenges Hearing Assistive Technol"},{"key":"ref3","doi-asserted-by":"crossref","first-page":"589","DOI":"10.1007\/s12559-019-09653-z","article-title":"A novel real-time, lightweight chaotic-encryption scheme for next-generation audio-visual hearing aids","volume":"12","author":"adeel","year":"2019","journal-title":"Cogn Comput"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2019.2917039"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2019.08.008"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2018.2889052"},{"key":"ref159","doi-asserted-by":"publisher","DOI":"10.1109\/JSSC.2007.914337"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1400"},{"key":"ref49","author":"deller","year":"2000","journal-title":"Discrete-Time Processing of Speech Signals"},{"key":"ref157","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054263"},{"key":"ref9","article-title":"LRS3-TED: A large-scale dataset for visual speech recognition","author":"afouras","year":"2018"},{"key":"ref158","doi-asserted-by":"publisher","DOI":"10.1186\/1687-6180-2012-183"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.1007\/s10844-016-0438-z"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1007\/BF02551274"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2064307"},{"key":"ref47","first-page":"32","article-title":"Audio-visual segmentation and &#x201C;the cocktail party effect","author":"darrell","year":"0","journal-title":"Proc Int Conf Multimodal Interfaces"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1065"},{"key":"ref41","first-page":"1","article-title":"Lip reading in profile","author":"chung","year":"0","journal-title":"Proc British Mach Vis Conf"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/34.927467"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1121\/1.2229005"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1121\/1.1358887"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1162\/089976600300015015"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.3109\/9781420088663"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00398"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2516"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.1016\/j.inffus.2020.04.001"},{"key":"ref74","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","author":"glorot","year":"0","journal-title":"Proc 13th Int Conf Artif Intell Statist"},{"key":"ref75","article-title":"AV speech enhancement challenge using a real noisy corpus","author":"gogate","year":"2019"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.1523\/JNEUROSCI.3675-12.2013"},{"key":"ref79","author":"goodfellow","year":"2016","journal-title":"Deep Learning"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1121\/1.1909702"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.213"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2014.2358871"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1097\/00003446-199204000-00003"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1002\/j.1538-7305.1929.tb01246.x"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462527"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1955"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01049"},{"key":"ref68","first-page":"35","article-title":"Learning to separate object sounds by watching unlabeled video","author":"gao","year":"0","journal-title":"Proc Eur Conf Comput Vis"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00041"},{"key":"ref197","first-page":"1","author":"ochiai","year":"0","journal-title":"Proc IEEE 27th Int Workshop Mach Learn Signal Process"},{"key":"ref198","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00772"},{"key":"ref199","first-page":"631","article-title":"Audio-visual scene analysis with self-supervised multisensory features","author":"owens","year":"0","journal-title":"Proc Eur Conf Comput Vis"},{"key":"ref193","doi-asserted-by":"publisher","DOI":"10.1080\/14992020903019312"},{"key":"ref194","doi-asserted-by":"publisher","DOI":"10.3109\/14992027.2010.524254"},{"key":"ref195","doi-asserted-by":"publisher","DOI":"10.1121\/1.408469"},{"key":"ref196","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1513"},{"key":"ref95","doi-asserted-by":"publisher","DOI":"10.3109\/14992027.2012.670731"},{"key":"ref94","doi-asserted-by":"publisher","DOI":"10.1186\/s13636-015-0054-9"},{"key":"ref190","doi-asserted-by":"publisher","DOI":"10.1049\/iet-spr.2011.0124"},{"key":"ref93","first-page":"1","article-title":"ViSQOL: The virtual speech quality objective listener","author":"hines","year":"0","journal-title":"Proc Int Workshop Acoust Signal Enhancement"},{"key":"ref191","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2010.2057198"},{"key":"ref92","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7471631"},{"key":"ref192","first-page":"689","article-title":"Multimodal deep learning","author":"ngiam","year":"0","journal-title":"Proc 28rd Int Conf Mach Learn"},{"key":"ref91","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.90"},{"key":"ref90","doi-asserted-by":"publisher","DOI":"10.1121\/1.1639908"},{"key":"ref98","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(89)90020-8"},{"key":"ref99","doi-asserted-by":"publisher","DOI":"10.1109\/TETCI.2017.2784878"},{"key":"ref96","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"ref97","doi-asserted-by":"publisher","DOI":"10.1016\/0893-6080(91)90009-T"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00633"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/TASSP.1984.1164317"},{"key":"ref84","article-title":"End-to-end multi-channel speech separation","author":"gu","year":"2019"},{"key":"ref83","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2266"},{"key":"ref80","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref89","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2407694"},{"key":"ref85","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2020.2980956"},{"key":"ref86","doi-asserted-by":"publisher","DOI":"10.3109\/01050398209076203"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.1080\/14992020500429583"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053215"},{"key":"ref200","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.264"},{"key":"ref101","doi-asserted-by":"publisher","DOI":"10.1121\/1.1909295"},{"key":"ref100","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPA.2016.7820732"},{"key":"ref209","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01381"},{"key":"ref203","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2017.8169995"},{"key":"ref204","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7951787"},{"key":"ref201","doi-asserted-by":"publisher","DOI":"10.3109\/14992021003681030"},{"key":"ref202","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462614"},{"key":"ref207","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054697"},{"key":"ref208","article-title":"CUAVE: A new audio-visual database for multimodal human-computer interface research","author":"patterson","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref205","doi-asserted-by":"publisher","DOI":"10.1109\/WASPAA.2019.8937237"},{"key":"ref206","doi-asserted-by":"crossref","first-page":"1272","DOI":"10.1126\/science.283.5406.1272","article-title":"Communication goes multimodal","volume":"283","author":"partan","year":"1999","journal-title":"Science"},{"key":"ref211","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1697"},{"key":"ref210","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952687"},{"key":"ref212","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2017.2738401"},{"key":"ref213","doi-asserted-by":"publisher","DOI":"10.1121\/1.1861713"},{"key":"ref214","author":"richie","year":"2009","journal-title":"Audiovisual Database of Spoken American English"},{"key":"ref215","doi-asserted-by":"publisher","DOI":"10.13053\/rcs-148-9-2"},{"key":"ref216","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.872619"},{"key":"ref217","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2007.04.008"},{"key":"ref218","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2013.2296173"},{"key":"ref219","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"ref220","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729586"},{"key":"ref222","first-page":"4492","article-title":"Ava active speaker: An audio-visual dataset for active speaker detection","author":"roth","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref221","first-page":"234","article-title":"U-Net: Convolutional networks for biomedical image segmentation","author":"ronneberger","year":"0","journal-title":"Proc Int Conf Med Image Comput Assist Interv"},{"key":"ref229","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2016.7846282"},{"key":"ref228","doi-asserted-by":"publisher","DOI":"10.1016\/S0031-3203(02)00031-6"},{"key":"ref227","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.3000593"},{"key":"ref226","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053730"},{"key":"ref225","article-title":"Mixture of inference networks for VAE-based audio-visual speech enhancement","author":"sadeghi","year":"2019"},{"key":"ref224","doi-asserted-by":"publisher","DOI":"10.1038\/323533a0"},{"key":"ref223","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682467"},{"key":"ref127","doi-asserted-by":"publisher","DOI":"10.1109\/ICSPCC.2014.6986284"},{"key":"ref126","first-page":"4485","article-title":"Transfer learning from speaker verification to multispeaker text-to-speech synthesis","author":"jia","year":"0","journal-title":"Proc 32nd Int Conf Neural Inf Process Syst"},{"key":"ref125","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2585878"},{"key":"ref124","author":"jekosch","year":"2006","journal-title":"Voice and Speech Quality Perception Assessment and Evaluation"},{"key":"ref129","first-page":"13286","article-title":"MMTM: Multimodal transfer module for CNN fusion","author":"joze","year":"0","journal-title":"Proc IEEE Conf Comput Vis and Pattern Recog"},{"key":"ref128","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.11.003"},{"key":"ref130","doi-asserted-by":"publisher","DOI":"10.1121\/1.381436"},{"key":"ref133","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2014.06.002"},{"key":"ref134","doi-asserted-by":"publisher","DOI":"10.17743\/jaes.2014.0006"},{"key":"ref131","doi-asserted-by":"publisher","DOI":"10.1121\/1.4784650"},{"key":"ref132","first-page":"363","article-title":"The hearing-aid speech quality index (HASQI)","volume":"58","author":"kates","year":"2010","journal-title":"J Audio Eng Soc"},{"key":"ref232","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref233","doi-asserted-by":"publisher","DOI":"10.1177\/1084713808325306"},{"key":"ref230","doi-asserted-by":"publisher","DOI":"10.1109\/78.650093"},{"key":"ref231","first-page":"1937","article-title":"Audio-visual scene analysis: Evidence for a &#x201C;very-early&#x201D; integration process in audio-visual speech perception","author":"schwartz","year":"0","journal-title":"Proc 7th Int Conf Spoken Lang Process - INTERSPEECH"},{"key":"ref239","doi-asserted-by":"publisher","DOI":"10.1121\/1.1907309"},{"key":"ref238","doi-asserted-by":"publisher","DOI":"10.1121\/1.1915893"},{"key":"ref235","article-title":"Conditioned source separation for music instrument performances","author":"slizovskaia","year":"2020"},{"key":"ref234","article-title":"Very deep convolutional networks for large-scale image recognition","author":"simonyan","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref237","doi-asserted-by":"publisher","DOI":"10.1155\/S1110865702207015"},{"key":"ref236","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2004.10.002"},{"key":"ref136","doi-asserted-by":"publisher","DOI":"10.1121\/1.1715112"},{"key":"ref135","doi-asserted-by":"publisher","DOI":"10.1016\/S0167-6393(98)00085-5"},{"key":"ref138","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2013.2261814"},{"key":"ref137","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2835719"},{"key":"ref139","doi-asserted-by":"publisher","DOI":"10.1214\/aoms\/1177729392"},{"key":"ref140","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-20873-8_18"},{"key":"ref141","first-page":"1755","article-title":"Dlib-ml: A machine learning toolkit","volume":"10","author":"king","year":"2009","journal-title":"J Mach Learn Res"},{"key":"ref142","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref143","doi-asserted-by":"publisher","DOI":"10.1121\/1.3179673"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1007\/s12559-013-9231-2"},{"key":"ref144","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2628641"},{"key":"ref1","first-page":"3752","article-title":"NTCD-TIMIT: A new database and baseline for noise-robust audio-visual speech recognition","author":"abdelaziz","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref145","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2726762"},{"key":"ref241","doi-asserted-by":"publisher","DOI":"10.1109\/HSCMA.2017.7895577"},{"key":"ref242","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-37734-2_60"},{"key":"ref243","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","author":"sutskever","year":"0","journal-title":"Proc 27th Int Conf Neural Inf Process Syst"},{"key":"ref244","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2114881"},{"key":"ref240","doi-asserted-by":"crossref","first-page":"71","DOI":"10.1098\/rstb.1992.0009","article-title":"Lipreading and audio-visual speech perception","volume":"335","author":"summerfield","year":"1992","journal-title":"Philos Trans Roy Soc London Ser B Biol Sci"},{"key":"ref248","first-page":"26","article-title":"Lecture 6.5 - RmsProp: Divide the gradient by a running average of its recent magnitude","volume":"4","author":"tieleman","year":"2012","journal-title":"COURSERA Neural Netw Mach Learn"},{"key":"ref247","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2020.2987209"},{"key":"ref246","article-title":"Exemplar-based lip-to-speech synthesis using convolutional neural networks","author":"takashima","year":"0","journal-title":"Proc IW-FCV"},{"key":"ref245","first-page":"1","article-title":"A survey on techniques for enhancing speech","volume":"179","author":"taha","year":"2018","journal-title":"Int J Comput Appl"},{"key":"ref249","article-title":"Detection and tracking of point features","author":"tomasi","year":"1991"},{"key":"ref109","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2671"},{"key":"ref108","doi-asserted-by":"publisher","DOI":"10.1109\/GlobalSIP45357.2019.8969244"},{"key":"ref107","article-title":"Audio-visual speech processing using deep learning techniques","author":"ideli","year":"2019"},{"key":"ref106","first-page":"29","article-title":"Towards multi-modal hearing aid design and evaluation in realistic audio-visual settings: Challenges and opportunities","author":"hussain","year":"0","journal-title":"Proc 1st Int Conf Challenges Hearing Assistive Technol"},{"key":"ref105","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2015.03.005"},{"key":"ref104","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.911054"},{"key":"ref103","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00397"},{"key":"ref102","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2018.00745"},{"key":"ref111","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","author":"ioffe","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref112","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1176"},{"key":"ref110","author":"institute","year":"1997","journal-title":"American National Standard Methods for Calculation of the Speech Intelligibility Index"},{"key":"ref250","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3269"},{"key":"ref251","doi-asserted-by":"publisher","DOI":"10.1080\/14992020500060875"},{"key":"ref254","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472692"},{"key":"ref255","doi-asserted-by":"publisher","DOI":"10.1109\/TSA.2005.858005"},{"key":"ref252","first-page":"6000","article-title":"Attention is all you need","author":"vaswani","year":"0","journal-title":"Proc 31st Int Conf Neural Inf Process Syst"},{"key":"ref253","doi-asserted-by":"publisher","DOI":"10.1007\/s10772-015-9295-3"},{"key":"ref257","doi-asserted-by":"publisher","DOI":"10.1023\/B:VISI.0000013087.49260.fb"},{"key":"ref256","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298935"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3114"},{"key":"ref259","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1445"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461856"},{"key":"ref258","first-page":"30","article-title":"Evaluating processed speech using the diagnostic rhyme test","volume":"1","author":"voiers","year":"1983"},{"key":"ref12","article-title":"Self-supervised learning of visual speech features with audiovisual speech enhancement","author":"aldeneh","year":"2020"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1121\/1.5042758"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2010.2096212"},{"key":"ref15","first-page":"2470","article-title":"Analysis of correlation between audio and visual speech features for clean audio feature prediction in noise","author":"almajai","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/FG.2015.7163155"},{"key":"ref118","year":"2001","journal-title":"Perceptual Evaluation of Speech Quality (PESQ) An Objective Method for End-to-End Speech Quality Assessment of Narrow-Band Telephone Networks and Speech Codecs"},{"key":"ref17","article-title":"Audio-visual target speaker extraction on multi-talker environment using event-driven cameras","author":"arriandiaga","year":"0","journal-title":"Proc IEEE Int Symp Circuits Syst"},{"key":"ref117","year":"1996","journal-title":"Subjective performance assessment of telephone-band and wideband digital codecs"},{"key":"ref18","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref19","first-page":"199","article-title":"Evidence of correlation between acoustic and visual features of speech","author":"barker","year":"0","journal-title":"Proc Int Congr Phonetic Sci"},{"key":"ref119","year":"2003","journal-title":"Subjective test methodology for evaluating speech communication systems that include noise suppression algorithm"},{"key":"ref114","year":"1998","journal-title":"Relative Timing of Sound and Vision for Broadcasting"},{"key":"ref113","year":"1990","journal-title":"Subjective Assessment of Sound Quality"},{"key":"ref116","year":"2019","journal-title":"General Methods for the Subjective Assessment of Sound Quality"},{"key":"ref115","year":"2003","journal-title":"Method for the Subjective Assessment of Intermediate Quality Levels of Coding Systems"},{"key":"ref120","year":"2003","journal-title":"Mapping Function for Transforming P 862 Raw Result Scores to MOS-LQO"},{"key":"ref121","year":"2005","journal-title":"Wideband Extension to Recommendation P 862 for the Assessment of Wideband Telephone Networks and Speech Codecs"},{"key":"ref122","year":"2011","journal-title":"Perceptual Objective Listening Quality Assessment"},{"key":"ref123","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054528"},{"key":"ref260","first-page":"44","article-title":"Entwicklung und evaluation eines satztests in deutscher sprache - Teil II: Optimierung des Oldenburger satztests","volume":"38","author":"wagener","year":"1999","journal-title":"Zeitschrift f&#x00FC;r Audiologie"},{"key":"ref261","first-page":"86","article-title":"Entwicklung und evaluation eines satztests in deutscher sprache - Teil III: Evaluierung des Oldenburger satztests","volume":"38","author":"wagener","year":"1999","journal-title":"Zeitschrift f&#x00FC;r Audiologie"},{"key":"ref262","first-page":"4","article-title":"Entwicklung und evaluation eines satztests in deutscher sprache - Teil I: Design des Oldenburger satztests","volume":"38","author":"wagener","year":"1999","journal-title":"Zeitschrift f&#x00FC;r Audiologie"},{"key":"ref263","author":"wang","year":"2006","journal-title":"Computational Auditory Scene Analysis Principles Algorithms and Applications"},{"key":"ref264","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2018.2842159"},{"key":"ref265","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1101"},{"key":"ref266","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053033"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9289074\/09380418.pdf?arnumber=9380418","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,3]],"date-time":"2024-12-03T19:09:48Z","timestamp":1733252988000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9380418\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021]]},"references-count":296,"URL":"https:\/\/doi.org\/10.1109\/taslp.2021.3066303","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021]]}}}