{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,14]],"date-time":"2026-07-14T14:58:08Z","timestamp":1784041088453,"version":"3.55.0"},"publisher-location":"New York, NY, USA","reference-count":48,"publisher":"ACM","license":[{"start":{"date-parts":[[2024,10,28]],"date-time":"2024-10-28T00:00:00Z","timestamp":1730073600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2024,10,28]]},"DOI":"10.1145\/3664647.3681261","type":"proceedings-article","created":{"date-parts":[[2024,10,26]],"date-time":"2024-10-26T06:59:33Z","timestamp":1729925973000},"page":"4748-4756","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":2,"title":["RAVSS: Robust Audio-Visual Speech Separation in Multi-Speaker Scenarios with Missing Visual Cues"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0002-8195-2005","authenticated-orcid":false,"given":"Tianrui","family":"Pan","sequence":"first","affiliation":[{"name":"State Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9297-7729","authenticated-orcid":false,"given":"Jie","family":"Liu","sequence":"additional","affiliation":[{"name":"State Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-6323-8989","authenticated-orcid":false,"given":"Bohan","family":"Wang","sequence":"additional","affiliation":[{"name":"State Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6086-3559","authenticated-orcid":false,"given":"Jie","family":"Tang","sequence":"additional","affiliation":[{"name":"State Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1391-1762","authenticated-orcid":false,"given":"Gangshan","family":"Wu","sequence":"additional","affiliation":[{"name":"State Key Laboratory for Novel Software Technology, Nanjing University, Nanjing, China"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2024,10,28]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Joon Son Chung, and Andrew Zisserman","author":"Afouras Triantafyllos","year":"2018","unstructured":"Triantafyllos Afouras, Joon Son Chung, and Andrew Zisserman. 2018. The Conversation: Deep Audio-Visual Speech Enhancement. arXiv:1804.04121 [cs.CV]"},{"key":"e_1_3_2_1_2_1","volume-title":"Joon Son Chung, and Andrew Zisserman","author":"Afouras Triantafyllos","year":"2018","unstructured":"Triantafyllos Afouras, Joon Son Chung, and Andrew Zisserman. 2018. LRS3-TED: a large-scale dataset for visual speech recognition. arXiv:1809.00496 [cs.CV]"},{"key":"e_1_3_2_1_3_1","volume-title":"Joon Son Chung, and Andrew Zisserman","author":"Afouras Triantafyllos","year":"2019","unstructured":"Triantafyllos Afouras, Joon Son Chung, and Andrew Zisserman. 2019. My lips are concealed: Audio-visual speech enhancement through obstructions. arXiv:1907.04975 [cs.CV]"},{"key":"e_1_3_2_1_4_1","volume-title":"The cocktail party phenomenon: A review of research on speech intelligibility in multiple-talker conditions. Acta acustica united with acustica 86, 1","author":"Bronkhorst Adelbert W","year":"2000","unstructured":"Adelbert W Bronkhorst. 2000. The cocktail party phenomenon: A review of research on speech intelligibility in multiple-talker conditions. Acta acustica united with acustica 86, 1 (2000), 117--128."},{"key":"e_1_3_2_1_5_1","unstructured":"Oscar Chang Otavio Braga Hank Liao Dmitriy Serdyuk and Olivier Siohan. On Robustness to Missing Video for Audiovisual Speech Recognition. Trans. Mach. Learn. Res. ([n. d.])."},{"key":"e_1_3_2_1_6_1","volume-title":"Continuous Speech Separation: Dataset and Analysis. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP. IEEE, 7284--7288","author":"Chen Zhuo","year":"2020","unstructured":"Zhuo Chen, Takuya Yoshioka, Liang Lu, Tianyan Zhou, Zhong Meng, Yi Luo, Jian Wu, Xiong Xiao, and Jinyu Li. 2020. Continuous Speech Separation: Dataset and Analysis. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP. IEEE, 7284--7288."},{"key":"e_1_3_2_1_7_1","volume-title":"Filter-Recovery Network for Multi-Speaker Audio-Visual Speech Separation. In International Conference on Learning Representations.","author":"Cheng Haoyue","year":"2023","unstructured":"Haoyue Cheng, Zhaoyang Liu, Wayne Wu, and LiminWang. 2023. Filter-Recovery Network for Multi-Speaker Audio-Visual Speech Separation. In International Conference on Learning Representations."},{"key":"e_1_3_2_1_8_1","doi-asserted-by":"publisher","DOI":"10.1121\/1.1907229"},{"key":"e_1_3_2_1_9_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1617"},{"key":"e_1_3_2_1_10_1","volume-title":"VoxCeleb2: Deep Speaker Recognition","author":"Chung Joon Son","unstructured":"Joon Son Chung, Arsha Nagrani, and Andrew Zisserman. 2018. VoxCeleb2: Deep Speaker Recognition. In International Speech Communication Association, B. Yegnanarayana (Ed.). ISCA, 1086--1090."},{"key":"e_1_3_2_1_11_1","unstructured":"Yusheng Dai Hang Chen Jun Du Ruoyu Wang Shihao Chen Jiefeng Ma Haotian Wang and Chin-Hui Lee. 2024. A Study of Dropout-Induced Modality Bias on Robustness to Missing Video Frames for Audio-Visual Speech Recognition. arXiv:2403.04245 [cs.SD]"},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"crossref","unstructured":"Shaked Dovrat Eliya Nachmani and Lior Wolf. 2021. Many-Speakers Single Channel Speech Separation with Optimal Permutation Training. arXiv:2104.08955 [cs.SD]","DOI":"10.21437\/Interspeech.2021-493"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.1145\/3197517.3201357"},{"key":"e_1_3_2_1_14_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR46437.2021.01524"},{"key":"e_1_3_2_1_15_1","volume-title":"The cocktail party problem. Neural computation 17, 9","author":"Haykin Simon","year":"2005","unstructured":"Simon Haykin and Zhe Chen. 2005. The cocktail party problem. Neural computation 17, 9 (2005), 1875--1902."},{"key":"e_1_3_2_1_16_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52729.2023.01801"},{"key":"e_1_3_2_1_17_1","volume-title":"Kingma and Jimmy Ba","author":"Diederik","year":"2017","unstructured":"Diederik P. Kingma and Jimmy Ba. 2017. Adam: A Method for Stochastic Optimization. arXiv:1412.6980 [cs.LG]"},{"key":"e_1_3_2_1_18_1","doi-asserted-by":"crossref","unstructured":"Morten Kolb\u00e6k Dong Yu Zheng-Hua Tan and Jesper Jensen. 2017. Multi-talker Speech Separation with Utterance-level Permutation Invariant Training of Deep Recurrent Neural Networks. arXiv:1703.06284 [cs.SD]","DOI":"10.1109\/TASLP.2017.2726762"},{"key":"e_1_3_2_1_19_1","unstructured":"Younglo Lee Shukjae Choi Byeong-Yeol Kim Zhong-Qiu Wang and Shinji Watanabe. 2024. Boosting Unknown-number Speaker Separation with Transformer Decoder-based Attractor. arXiv:2401.12473 [eess.AS]"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2024.3375641"},{"key":"e_1_3_2_1_21_1","volume-title":"Av-Sepformer: Cross-Attention Sepformer for Audio-Visual Target Speaker Extraction. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 1--5.","author":"Lin Jiuxin","year":"2023","unstructured":"Jiuxin Lin, Xinyu Cai, Heinrich Dinkel, Jun Chen, Zhiyong Yan, Yongqing Wang, Junbo Zhang, Zhiyong Wu, Yujun Wang, and Helen Meng. 2023. Av-Sepformer: Cross-Attention Sepformer for Audio-Visual Target Speaker Extraction. In IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP). 1--5."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"crossref","unstructured":"Yi Luo Zhuo Chen and Takuya Yoshioka. 2020. Dual-path RNN: efficient long sequence modeling for time-domain single-channel speech separation. arXiv:1910.06379 [eess.AS]","DOI":"10.1109\/ICASSP40776.2020.9054266"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.1109\/taslp.2019.2915167"},{"key":"e_1_3_2_1_24_1","volume-title":"Sepit: Approaching a single channel speech separation bound. arXiv preprint arXiv:2205.11801","author":"Lutati Shahar","year":"2022","unstructured":"Shahar Lutati, Eliya Nachmani, and Lior Wolf. 2022. Sepit: Approaching a single channel speech separation bound. arXiv preprint arXiv:2205.11801 (2022)."},{"key":"e_1_3_2_1_25_1","unstructured":"Shahar Lutati Eliya Nachmani and Lior Wolf. 2023. Separate And Diffuse: Using a Pretrained Diffusion Model for Improving Source Separation. arXiv:2301.10752 [eess.AS]"},{"key":"e_1_3_2_1_26_1","doi-asserted-by":"crossref","unstructured":"Naoki Makishima Mana Ihori Akihiko Takashima Tomohiro Tanaka Shota Orihashi and Ryo Masumura. 2021. Audio-Visual Speech Separation Using Cross-Modal Correspondence Loss. arXiv:2103.01463 [cs.SD]","DOI":"10.1109\/ICASSP39728.2021.9413491"},{"key":"e_1_3_2_1_27_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1753"},{"key":"e_1_3_2_1_28_1","doi-asserted-by":"crossref","unstructured":"H\u00e9ctor Martel Julius Richter Kai Li Xiaolin Hu and Timo Gerkmann. 2023. Audio-Visual Speech Separation in Noisy Environments with a Lightweight Iterative Model. arXiv:2306.00160 [eess.AS] https:\/\/arxiv.org\/abs\/2306.00160","DOI":"10.21437\/Interspeech.2023-1753"},{"key":"e_1_3_2_1_29_1","volume-title":"d.]. Self-Supervised Speech Representation Learning: A Review","author":"Mohamed Abdelrahman","unstructured":"Abdelrahman Mohamed, Hung-yi Lee, Lasse Borgholt, Jakob D. Havtorn, Joakim Edin, Christian Igel, Katrin Kirchhoff, Shang-Wen Li, Karen Livescu, Lars Maaloe, Tara N. Sainath, and Shinji Watanabe. [n. d.]. Self-Supervised Speech Representation Learning: A Review. IEEE Journal of Selected Topics in Signal Processing ([n. d.])."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"crossref","unstructured":"Juan F. Montesinos Venkatesh S. Kadandale and Gloria Haro. 2022. VoViT: Low Latency Graph-based Audio-Visual Voice Separation Transformer. arXiv:2203.04099 [cs.SD] https:\/\/arxiv.org\/abs\/2203.04099","DOI":"10.1007\/978-3-031-19836-6_18"},{"key":"e_1_3_2_1_31_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3205759"},{"key":"e_1_3_2_1_32_1","volume-title":"Muse: Multi-Modal Target Speaker Extraction with Visual Cues. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2021","author":"Pan Zexu","year":"2021","unstructured":"Zexu Pan, Ruijie Tao, Chenglin Xu, and Haizhou Li. 2021. Muse: Multi-Modal Target Speaker Extraction with Visual Cues. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP 2021, Toronto, ON, Canada, June 6-11, 2021. IEEE, 6678--6682."},{"key":"e_1_3_2_1_33_1","unstructured":"Samuel Pegg Kai Li and Xiaolin Hu. 2024. RTFS-Net: Recurrent time-frequency modelling for efficient audio-visual speech separation. arXiv:2309.17189 [cs.SD]"},{"key":"e_1_3_2_1_34_1","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.01024"},{"key":"e_1_3_2_1_35_1","volume-title":"IEEE Spoken Language Technology Workshop SLT. IEEE, 897--904","author":"Raj Desh","unstructured":"Desh Raj, Pavel Denisov, Zhuo Chen, Hakan Erdogan, Zili Huang, Maokui He, Shinji Watanabe, Jun Du, Takuya Yoshioka, Yi Luo, Naoyuki Kanda, Jinyu Li, Scott Wisdom, and John R. Hershey. 2021. Integration of Speech Separation, Diarization, and Recognition for Multi-Speaker Meetings: System Description, Comparison, and Analysis. In IEEE Spoken Language Technology Workshop SLT. IEEE, 897--904."},{"key":"e_1_3_2_1_36_1","volume-title":"2001 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No.01CH37221)","volume":"2","author":"Rix A.W.","unstructured":"A.W. Rix, J.G. Beerends, M.P. Hollier, and A.P. Hekstra. 2001. Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs. In 2001 IEEE International Conference on Acoustics, Speech, and Signal Processing. Proceedings (Cat. No.01CH37221), Vol. 2. 749--752 vol.2."},{"key":"e_1_3_2_1_37_1","volume-title":"Hershey","author":"Roux Jonathan Le","year":"2018","unstructured":"Jonathan Le Roux, Scott Wisdom, Hakan Erdogan, and John R. Hershey. 2018. SDR - half-baked or well done? arXiv:1811.02508 [cs.SD]"},{"key":"e_1_3_2_1_38_1","unstructured":"Bowen Shi Wei-Ning Hsu Kushal Lakhotia and Abdelrahman Mohamed. 2022. Learning Audio-Visual Speech Representation by Masked Multimodal Cluster Prediction. arXiv:2201.02184 [eess.AS]"},{"key":"e_1_3_2_1_39_1","doi-asserted-by":"crossref","unstructured":"Cem Subakan Mirco Ravanelli Samuele Cornell Mirko Bronzi and Jianyuan Zhong. 2021. Attention is All You Need in Speech Separation. arXiv:2010.13154 [eess.AS]","DOI":"10.1109\/ICASSP39728.2021.9413901"},{"key":"e_1_3_2_1_40_1","volume-title":"Attention is all you need. Advances in neural information processing systems 30","author":"Vaswani Ashish","year":"2017","unstructured":"Ashish Vaswani, Noam Shazeer, Niki Parmar, Jakob Uszkoreit, Llion Jones, Aidan N Gomez, \u0141ukasz Kaiser, and Illia Polosukhin. 2017. Attention is all you need. Advances in neural information processing systems 30 (2017)."},{"key":"e_1_3_2_1_41_1","doi-asserted-by":"publisher","DOI":"10.1109\/MSPEC.2017.7864754"},{"key":"e_1_3_2_1_42_1","doi-asserted-by":"publisher","DOI":"10.1609\/aaai.v38i17.29882"},{"key":"e_1_3_2_1_43_1","unstructured":"Zhong-Qiu Wang Samuele Cornell Shukjae Choi Younglo Lee Byeong-Yeol Kim and Shinji Watanabe. 2023. TF-GridNet: Making Time-Frequency Domain Models Great Again for Monaural Speaker Separation. arXiv:2209.03952 [cs.SD]"},{"key":"e_1_3_2_1_44_1","volume-title":"Time Domain Audio Visual Speech Separation. In IEEE Automatic Speech Recognition and Understanding Workshop, ASRU. IEEE, 667--673","author":"Wu Jian","year":"2019","unstructured":"Jian Wu, Yong Xu, Shi-Xiong Zhang, Lianwu Chen, Meng Yu, Lei Xie, and Dong Yu. 2019. Time Domain Audio Visual Speech Separation. In IEEE Automatic Speech Recognition and Understanding Workshop, ASRU. IEEE, 667--673."},{"key":"e_1_3_2_1_45_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2019.00097"},{"key":"e_1_3_2_1_46_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747554"},{"key":"e_1_3_2_1_47_1","volume-title":"d.]. Stepwise-Refining Speech Separation Network via Fine-Grained Encoding in High-Order Latent Domain","author":"Yao Zengwei","unstructured":"Zengwei Yao, Wenjie Pei, Fanglin Chen, Guangming Lu, and David Zhang. [n. d.]. Stepwise-Refining Speech Separation Network via Fine-Grained Encoding in High-Order Latent Domain. IEEE ACM Trans. Audio Speech Lang. Process. ([n. d.])."},{"key":"e_1_3_2_1_48_1","volume-title":"IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP.","author":"Yu Dong","unstructured":"Dong Yu, Morten Kolb\u00e6k, Zheng-Hua Tan, and Jesper Jensen. [n. d.]. Permutation invariant training of deep models for speaker-independent multi-talker speech separation. In IEEE International Conference on Acoustics, Speech and Signal Processing, ICASSP."}],"event":{"name":"MM '24: The 32nd ACM International Conference on Multimedia","location":"Melbourne VIC Australia","acronym":"MM '24","sponsor":["SIGMM ACM Special Interest Group on Multimedia"]},"container-title":["Proceedings of the 32nd ACM International Conference on Multimedia"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681261","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3664647.3681261","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,19]],"date-time":"2025-06-19T01:17:42Z","timestamp":1750295862000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3664647.3681261"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,28]]},"references-count":48,"alternative-id":["10.1145\/3664647.3681261","10.1145\/3664647"],"URL":"https:\/\/doi.org\/10.1145\/3664647.3681261","relation":{},"subject":[],"published":{"date-parts":[[2024,10,28]]},"assertion":[{"value":"2024-10-28","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}