{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:57:56Z","timestamp":1785488276515,"version":"3.56.0"},"publisher-location":"New York, NY, USA","reference-count":36,"publisher":"ACM","license":[{"start":{"date-parts":[[2025,12,17]],"date-time":"2025-12-17T00:00:00Z","timestamp":1765929600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2025,12,17]]},"DOI":"10.1145\/3774521.3774597","type":"proceedings-article","created":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T07:34:24Z","timestamp":1785483264000},"page":"1-9","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":0,"title":["Low SNR Speech Perception with HuBERT: A Discussion on Visual-Audio Fusion and Domain-Specific Modeling"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-0839-1849","authenticated-orcid":false,"given":"Jayasree","family":"Saha","sequence":"first","affiliation":[{"name":"UPES, Bidholii Campus, Dehradun, India"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-5262-9722","authenticated-orcid":false,"given":"vinay","family":"Namboodiri","sequence":"additional","affiliation":[{"name":"University of Bath, Bath, United Kingdom"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6767-7057","authenticated-orcid":false,"given":"C. V","family":"Jawahar","sequence":"additional","affiliation":[{"name":"IIIT hyderabad, Hyderabad, India"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2026,7,31]]},"reference":[{"key":"e_1_3_3_1_2_2","doi-asserted-by":"crossref","unstructured":"Ahsan Adeel Mandar Gogate and Amir Hussain. 2020. Contextual deep learning-based audio-visual switching for speech enhancement in real-world environments. Information Fusion 59 (2020) 163\u2013170.","DOI":"10.1016\/j.inffus.2019.08.008"},{"key":"e_1_3_3_1_3_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3114"},{"key":"e_1_3_3_1_4_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054224"},{"key":"e_1_3_3_1_5_2","unstructured":"Alexei Baevski Yuhao Zhou Abdelrahman Mohamed and Michael Auli. 2020. wav2vec 2.0: A framework for self-supervised learning of speech representations. Advances in Neural Information Processing Systems 33 (2020)."},{"key":"e_1_3_3_1_6_2","unstructured":"Dan Biderman Jose\u00a0Gonzalez Ortiz Jacob Portes Mansheej Paul Philip Greengard Connor Jennings Daniel King Sam Havens Vitaliy Chiley Jonathan Frankle Cody Blakeney and John\u00a0P. Cunningham. 2024. LoRA Learns Less and Forgets Less. arxiv:https:\/\/arXiv.org\/abs\/2405.09673\u00a0[cs.LG]"},{"key":"e_1_3_3_1_7_2","doi-asserted-by":"crossref","unstructured":"Anyuan Chen Chengyi Wang Zhengyang Chen Yu Wu Shujie Liu Zhuo Chen Jinyu Li Naoyuki Kanda Takuya Yoshioka Xiong Xiao et\u00a0al. 2022. Wavlm: Large-scale self-supervised pre-training for full stack speech processing. IEEE Journal of Selected Topics in Signal Processing 16 6 (2022) 1505\u20131518.","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"e_1_3_3_1_8_2","doi-asserted-by":"crossref","unstructured":"F. Chen Y. Hu and M. Yuan. 2015. Evaluation of noise reduction methods for sentence recognition by mandarin-speaking cochlear implant listeners. Ear Hear. 36 1 (2015) 61\u201371.","DOI":"10.1097\/AUD.0000000000000074"},{"key":"e_1_3_3_1_9_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSPW59220.2023.10193049"},{"key":"e_1_3_3_1_10_2","doi-asserted-by":"publisher","DOI":"10.1201\/9781315220109"},{"key":"e_1_3_3_1_11_2","volume-title":"Interspeech","author":"Defossez Alexandre","year":"2020","unstructured":"Alexandre Defossez, Gabriel Synnaeve, and Yossi Adi. 2020. Real Time Speech Enhancement in the Waveform Domain. In Interspeech."},{"key":"e_1_3_3_1_12_2","doi-asserted-by":"publisher","DOI":"10.5555\/1337690.1338318"},{"key":"e_1_3_3_1_13_2","doi-asserted-by":"crossref","unstructured":"Mandar Gogate Kia Dashtipour Ahsan Adeel and Amir Hussain. 2020. CochleaNet: A robust language-independent audio-visual model for real-time speech enhancement. Information Fusion 63 (2020) 273\u2013285.","DOI":"10.1016\/j.inffus.2020.04.001"},{"key":"e_1_3_3_1_14_2","doi-asserted-by":"crossref","unstructured":"E.\u00a0W. Healy M. Delfarah Eric\u00a0M. Johnson and DeLiang Wang. 2019. A deep learning algorithm to increase intelligibility for hearing-impaired listeners in the presence of a competing talker and reverberation. J. Acoustical Soc. Amer. 145 3 (2019) 1378\u20131388.","DOI":"10.1121\/1.5093547"},{"key":"e_1_3_3_1_15_2","doi-asserted-by":"publisher","DOI":"10.1109\/WACV48630.2021.00197"},{"key":"e_1_3_3_1_16_2","doi-asserted-by":"crossref","unstructured":"Wei-Ning Hsu et\u00a0al. 2021. Hubert: Self-supervised speech representation learning by masked prediction of hidden units. IEEE\/ACM Transactions on Audio Speech and Language Processing 29 (2021) 3451\u20133460.","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_3_1_17_2","unstructured":"Edward\u00a0J. Hu Yelong Shen Phillip Wallis Zeyuan Allen-Zhu Yuanzhi Li Shean Wang Lu Wang and Weizhu Chen. 2021. LoRA: Low-Rank Adaptation of Large Language Models. arxiv:https:\/\/arXiv.org\/abs\/2106.09685"},{"key":"e_1_3_3_1_18_2","volume-title":"International Conference on Learning Representations","author":"Hu Edward\u00a0J","year":"2022","unstructured":"Edward\u00a0J Hu, Yelong Shen, Phillip Wallis, Zeyuan Allen-Zhu, Yuanzhi Li, Shean Wang, Lu Wang, and Weizhu Chen. 2022. LoRA: Low-Rank Adaptation of Large Language Models. In International Conference on Learning Representations. https:\/\/openreview.net\/forum?id=nZeVKeeFYf9"},{"key":"e_1_3_3_1_19_2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2858"},{"key":"e_1_3_3_1_20_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414922"},{"key":"e_1_3_3_1_21_2","unstructured":"H. Levit. 2001. Noise reduction in hearing aids: An overview. J. Rehabil. Res. Develop. 38 1 (2001) 111\u2013121."},{"key":"e_1_3_3_1_22_2","volume-title":"Robust Automatic Speech Recognition: A Bridge to Practical Applications","author":"Li J.","year":"2015","unstructured":"J. Li, L. Deng, R. Haeb-Umbach, and Y. Gong. 2015. Robust Automatic Speech Recognition: A Bridge to Practical Applications. Academic Press, New York, NY, USA."},{"key":"e_1_3_3_1_23_2","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.68"},{"key":"e_1_3_3_1_24_2","unstructured":"Xiang\u00a0Lisa Li and Percy Liang. 2021. Prefix-tuning: Optimizing continuous prompts for generation. arXiv preprint arXiv:https:\/\/arXiv.org\/abs\/2101.00190 (2021)."},{"key":"e_1_3_3_1_25_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447004"},{"key":"e_1_3_3_1_26_2","doi-asserted-by":"crossref","unstructured":"Daniel Michelsanti Zheng-Hua Tan Sigurdur Sigurdsson and Jesper Jensen. 2019. Deep-learning-based audio-visual speech enhancement in presence of Lombard effect. Speech Communication 115 (2019) 38\u201350.","DOI":"10.1016\/j.specom.2019.10.006"},{"key":"e_1_3_3_1_27_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICSLP.1996.607754"},{"key":"e_1_3_3_1_28_2","doi-asserted-by":"crossref","unstructured":"M.\u00a0L. Patterson and J.\u00a0F Werker. 2003. Two-month-old infants match phonetic information in lips and voice. Developmental Science 6 (2003) 191\u2013196.","DOI":"10.1111\/1467-7687.00271"},{"key":"e_1_3_3_1_29_2","doi-asserted-by":"crossref","unstructured":"Summerfield Q. 1979. Use of visual information for phonetic perception. Phonetica 34 (1979) 314\u2013331.","DOI":"10.1159\/000259969"},{"key":"e_1_3_3_1_30_2","unstructured":"Sylvestre-Alvise Rebuffi et\u00a0al. 2017. Learning multiple visual domains with residual adapters. Advances in Neural Information Processing Systems 30 (2017)."},{"key":"e_1_3_3_1_31_2","doi-asserted-by":"crossref","unstructured":"W.\u00a0H. Sumby and I. Pollack. 1954. Visual contribution to speech intelligibility in noise. Journal of the Acoustical Society of America 26 (1954) 212\u2013215.","DOI":"10.1121\/1.1907309"},{"key":"e_1_3_3_1_32_2","doi-asserted-by":"crossref","unstructured":"Wenxin Tai. 2022. A Repetitive Spectrum Learning Framework for Monaural Speech Enhancement in Extremely Low SNR Environments (Student Abstract). Proceedings of the AAAI Conference on Artificial Intelligence 36 11 (2022) 13063\u201313064.","DOI":"10.1609\/aaai.v36i11.21668"},{"key":"e_1_3_3_1_33_2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746223"},{"key":"e_1_3_3_1_34_2","doi-asserted-by":"crossref","unstructured":"Steven Vander\u00a0Eeckt et\u00a0al. 2022. Using adapters to overcome catastrophic forgetting in end-to-end automatic speech recognition. arXiv e-prints (2022) arXiv\u20132203.","DOI":"10.23919\/EUSIPCO55093.2022.9909589"},{"key":"e_1_3_3_1_35_2","doi-asserted-by":"publisher","DOI":"10.5555\/3307050"},{"key":"e_1_3_3_1_36_2","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR52688.2022.00805"},{"key":"e_1_3_3_1_37_2","volume-title":"International Conference on Learning Representations","author":"Zhong Zihan","year":"2024","unstructured":"Zihan Zhong, Zhiqiang Tang, Tong He, Haoyang Fang, and Chun Yuan. 2024. Convolution Meets LoRA: Parameter Efficient Finetuning for Segment Anything Model. In International Conference on Learning Representations."}],"event":{"name":"ICVGIP 2025: Indian Conference on Computer Vision, Graphics, and Image Processing","location":"Mandi Himachal Pradesh India","acronym":"ICVGIP 2025"},"container-title":["Proceedings of the Sixteen Indian Conference on Computer Vision, Graphics and Image Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3774521.3774597","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,7,31]],"date-time":"2026-07-31T08:03:34Z","timestamp":1785485014000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3774521.3774597"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,12,17]]},"references-count":36,"alternative-id":["10.1145\/3774521.3774597","10.1145\/3774521"],"URL":"https:\/\/doi.org\/10.1145\/3774521.3774597","relation":{},"subject":[],"published":{"date-parts":[[2025,12,17]]},"assertion":[{"value":"2026-07-31","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}