{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,6]],"date-time":"2025-11-06T12:28:21Z","timestamp":1762432101974,"version":"3.28.0"},"reference-count":50,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,7,18]]},"DOI":"10.1109\/ijcnn52387.2021.9533660","type":"proceedings-article","created":{"date-parts":[[2021,9,20]],"date-time":"2021-09-20T21:27:41Z","timestamp":1632173261000},"page":"1-8","source":"Crossref","is-referenced-by-count":2,"title":["Audio-Visual Speech Separation with Visual Features Enhanced by Adversarial Training"],"prefix":"10.1109","author":[{"given":"Peng","family":"Zhang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiaming","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jing","family":"Shi","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yunzhe","family":"Hao","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Lei","family":"Qin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Bo","family":"Xu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"doi-asserted-by":"publisher","key":"ref39","DOI":"10.1007\/978-3-319-54184-6_6"},{"doi-asserted-by":"publisher","key":"ref38","DOI":"10.21437\/Interspeech.2020-2085"},{"doi-asserted-by":"publisher","key":"ref33","DOI":"10.1109\/ICASSP.2018.8462116"},{"doi-asserted-by":"publisher","key":"ref32","DOI":"10.1109\/ASRU46091.2019.9003983"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.21437\/Interspeech.2018-1955"},{"doi-asserted-by":"publisher","key":"ref30","DOI":"10.1109\/ICASSP.2018.8462527"},{"doi-asserted-by":"publisher","key":"ref37","DOI":"10.1109\/TASLP.2019.2915167"},{"doi-asserted-by":"publisher","key":"ref36","DOI":"10.1007\/978-3-319-54427-4_19"},{"key":"ref35","article-title":"Least Squares Generative Adversarial Networks","author":"mao","year":"2016","journal-title":"ArXiv Preprint"},{"doi-asserted-by":"publisher","key":"ref34","DOI":"10.1007\/978-3-030-01228-1_5"},{"doi-asserted-by":"publisher","key":"ref28","DOI":"10.1121\/1.1358887"},{"doi-asserted-by":"publisher","key":"ref27","DOI":"10.1109\/ICASSP.2005.1416331"},{"key":"ref29","first-page":"1173","article-title":"Audio-visual sound separation via hidden Markov models","author":"hershey","year":"2002","journal-title":"Advances in neural information processing systems"},{"doi-asserted-by":"publisher","key":"ref2","DOI":"10.1016\/j.cub.2009.09.005"},{"doi-asserted-by":"publisher","key":"ref1","DOI":"10.1121\/1.1907229"},{"doi-asserted-by":"publisher","key":"ref20","DOI":"10.1109\/TASLP.2019.2928140"},{"doi-asserted-by":"publisher","key":"ref22","DOI":"10.1109\/ICASSP.2019.8682061"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.1109\/LSP.2018.2853566"},{"key":"ref24","article-title":"Multimodal Target Speech Separation with Voice and Face References","author":"qu","year":"0","journal-title":"Proc of Interspeech"},{"doi-asserted-by":"publisher","key":"ref23","DOI":"10.21437\/Interspeech.2020-1065"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1609\/aaai.v33i01.33019299"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1109\/ICASSP.2017.7953127"},{"doi-asserted-by":"publisher","key":"ref50","DOI":"10.1109\/ICASSP.2017.7952261"},{"year":"2014","author":"massaro","journal-title":"Speech Perception by Ear and Eye A Paradigm for Psychological Inquiry","key":"ref10"},{"key":"ref11","first-page":"1755","article-title":"Dlib-ml: A machine learning toolkit","volume":"10","author":"king","year":"2009","journal-title":"Journal of Machine Learning Research"},{"doi-asserted-by":"publisher","key":"ref40","DOI":"10.1007\/978-3-319-46487-9_6"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1109\/LSP.2016.2603342"},{"key":"ref13","article-title":"An iterative image registration technique with an application to stereo vision","author":"lucas","year":"0","journal-title":"Proc of IJCAI 1981"},{"key":"ref14","article-title":"Detection and tracking of point features","author":"tomasi","year":"1991","journal-title":"Technical Report CMU-CS-91&#x2013;132"},{"doi-asserted-by":"publisher","key":"ref15","DOI":"10.1145\/3197517.3201357"},{"key":"ref16","article-title":"Muse: Multi-modal target speaker extraction with visual cues","author":"pan","year":"2020","journal-title":"ArXiv Preprint"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.21437\/Interspeech.2019-3114"},{"doi-asserted-by":"publisher","key":"ref18","DOI":"10.21437\/Interspeech.2018-1400"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.1109\/JSTSP.2020.2987209"},{"key":"ref4","doi-asserted-by":"crossref","first-page":"1901","DOI":"10.1109\/TASLP.2017.2726762","article-title":"Multitalker Speech Separation With Utterance-Level Permutation Invariant Training of Deep Recurrent Neural Networks","volume":"25","author":"morten","year":"2017","journal-title":"IEEE\/ACM Transactions on Audio Speech and Language Processing"},{"doi-asserted-by":"publisher","key":"ref3","DOI":"10.1109\/ICASSP.2017.7952154"},{"key":"ref6","first-page":"7164","article-title":"Voice Separation with an Unknown Number of Multiple Speakers","author":"nachmani","year":"0","journal-title":"Proc of ICML"},{"key":"ref5","article-title":"Single Channel auditory source separation with neural network","author":"chen","year":"2017","journal-title":"Columbia University"},{"doi-asserted-by":"publisher","key":"ref8","DOI":"10.1038\/264746a0"},{"doi-asserted-by":"publisher","key":"ref7","DOI":"10.1016\/j.actpsy.2010.03.010"},{"doi-asserted-by":"publisher","key":"ref49","DOI":"10.1109\/TASL.2011.2114881"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.3109\/03005368709077786"},{"doi-asserted-by":"publisher","key":"ref46","DOI":"10.1109\/TSA.2005.858005"},{"doi-asserted-by":"publisher","key":"ref45","DOI":"10.21437\/Interspeech.2016-1176"},{"doi-asserted-by":"publisher","key":"ref48","DOI":"10.1609\/aaai.v34i05.6489"},{"key":"ref47","first-page":"749","article-title":"Perceptual evaluation of speech quality (pesq)-a new method for speech quality assessment of telephone networks and codecs","volume":"2","author":"antony","year":"0","journal-title":"Proc of ICASSP"},{"doi-asserted-by":"publisher","key":"ref42","DOI":"10.1109\/TMM.2015.2407694"},{"doi-asserted-by":"publisher","key":"ref41","DOI":"10.1121\/1.2229005"},{"key":"ref44","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"Proc Int Conf Lear Represent"},{"key":"ref43","article-title":"Deep audiovisual speech recognition","author":"afouras","year":"2018","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"}],"event":{"name":"2021 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2021,7,18]]},"location":"Shenzhen, China","end":{"date-parts":[[2021,7,22]]}},"container-title":["2021 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9533266\/9533267\/09533660.pdf?arnumber=9533660","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:46:02Z","timestamp":1652197562000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9533660\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,18]]},"references-count":50,"URL":"https:\/\/doi.org\/10.1109\/ijcnn52387.2021.9533660","relation":{},"subject":[],"published":{"date-parts":[[2021,7,18]]}}}