{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,18]],"date-time":"2026-07-18T18:32:35Z","timestamp":1784399555264,"version":"3.55.0"},"reference-count":27,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,6,6]],"date-time":"2021-06-06T00:00:00Z","timestamp":1622937600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2021,6,6]],"date-time":"2021-06-06T00:00:00Z","timestamp":1622937600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,6,6]],"date-time":"2021-06-06T00:00:00Z","timestamp":1622937600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100001321","name":"National Research Foundation","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001321","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001659","name":"Deutsche Forschungsgemeinschaft","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100001659","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,6,6]]},"DOI":"10.1109\/icassp39728.2021.9414023","type":"proceedings-article","created":{"date-parts":[[2021,5,13]],"date-time":"2021-05-13T19:53:45Z","timestamp":1620935625000},"page":"6678-6682","source":"Crossref","is-referenced-by-count":40,"title":["Muse: Multi-Modal Target Speaker Extraction with Visual Cues"],"prefix":"10.1109","author":[{"given":"Zexu","family":"Pan","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Ruijie","family":"Tao","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Chenglin","family":"Xu","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Haizhou","family":"Li","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1397"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3197517.3201357","article-title":"Looking to listen at the cocktail party: a speaker-independent audio-visual model for speech separation","volume":"37","author":"ephrat","year":"2018","journal-title":"ACM Transactions on Graphics"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1400"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003983"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1513"},{"key":"ref15","first-page":"4492","article-title":"AVA active speaker: An audio-visual dataset for active speaker detection","author":"roth","year":"2020","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3114"},{"key":"ref17","article-title":"Theory and application of digital signal processing","author":"oppenheim","year":"1978","journal-title":"Englewood Cliffs"},{"key":"ref18","article-title":"Deep audio-visual speech recognition","author":"afouras","year":"2018","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"ref19","first-page":"626","article-title":"SDR&#x2013;half-baked or well done?","author":"le roux","year":"2019","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref4","first-page":"31","article-title":"Deep clustering: Discriminative embeddings for segmentation and separation","author":"hershey","year":"2016","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01231-1_39"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.885253"},{"key":"ref6","first-page":"6","article-title":"Single channel speech separation with constrained utterance level permutation invariant training using grid lstm","author":"xu","year":"2018","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2915167"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1101"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2020.2987429"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2006.881700"},{"key":"ref9","first-page":"376","article-title":"Speakerfilter: Deep learning-based target speaker extraction using anchor speech","author":"he","year":"2020","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref1","first-page":"117","article-title":"The cocktail party phenomenon: A review of research on speech intelligibility in multiple-talker conditions","volume":"86","author":"bronkhorst","year":"2000","journal-title":"Acta Acustica United with Acustica"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/TMM.2015.2407694"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1121\/1.2229005"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2001.941023"},{"key":"ref23","article-title":"LRS3-TED: a large-scale dataset for visual speech recognition","author":"afouras","year":"2018"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1065"},{"key":"ref25","first-page":"276","article-title":"Audio-visual speech separation using i-vectors","author":"yiyu","year":"2019","journal-title":"IEEE Int Conf Information Communication Signal Processing"}],"event":{"name":"ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Toronto, ON, Canada","start":{"date-parts":[[2021,6,6]]},"end":{"date-parts":[[2021,6,11]]}},"container-title":["ICASSP 2021 - 2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9413349\/9413350\/09414023.pdf?arnumber=9414023","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,10]],"date-time":"2022-05-10T15:40:51Z","timestamp":1652197251000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9414023\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,6,6]]},"references-count":27,"URL":"https:\/\/doi.org\/10.1109\/icassp39728.2021.9414023","relation":{},"subject":[],"published":{"date-parts":[[2021,6,6]]}}}