{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T20:13:19Z","timestamp":1783800799868,"version":"3.55.0"},"reference-count":63,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2027,2,1]],"date-time":"2027-02-01T00:00:00Z","timestamp":1801440000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T00:00:00Z","timestamp":1783382400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by\/4.0\/"}],"funder":[{"DOI":"10.13039\/501100000038","name":"Natural Sciences and Engineering Research Council of Canada","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100000038","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2027,2]]},"DOI":"10.1016\/j.csl.2026.102023","type":"journal-article","created":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T16:03:31Z","timestamp":1783440211000},"page":"102023","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Robust audio\u2013visual speech enhancement under visual occlusion using TITR-CNN and silence-aware MALA-EM"],"prefix":"10.1016","volume":"102","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-2801-7723","authenticated-orcid":false,"given":"Z.","family":"Foroushi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"R.M.","family":"Dansereau","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.102023_b1","series-title":"Proc. Interspeech 2017","first-page":"3752","article-title":"NTCD-TIMIT: A new database and baseline for noise-robust audio-visual speech recognition","author":"Abdelaziz","year":"2017"},{"key":"10.1016\/j.csl.2026.102023_b2","doi-asserted-by":"crossref","unstructured":"Afouras, T., Chung, J.S., Zisserman, A., 2018. The Conversation: Deep Audio\u2013Visual Speech Enhancement. In: Proceedings of Interspeech. Hyderabad, India, pp. 3244\u20133248. http:\/\/dx.doi.org\/10.48550\/arXiv.1804.04121.","DOI":"10.21437\/Interspeech.2018-1400"},{"key":"10.1016\/j.csl.2026.102023_b3","series-title":"Proceedings of the Annual Conference of the International Speech Communication Association","first-page":"4295","article-title":"My lips are concealed: Audio-visual speech enhancement through obstructions","author":"Afouras","year":"2019"},{"key":"10.1016\/j.csl.2026.102023_b4","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"716","article-title":"Statistical speech enhancement based on probabilistic integration of variational autoencoder and non-negative matrix factorization","author":"Bando","year":"2018"},{"key":"10.1016\/j.csl.2026.102023_b5","doi-asserted-by":"crossref","first-page":"2993","DOI":"10.1109\/TASLP.2022.3207349","article-title":"Unsupervised speech enhancement using dynamical variational autoencoders","volume":"30","author":"Bie","year":"2022","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"2","key":"10.1016\/j.csl.2026.102023_b6","doi-asserted-by":"crossref","first-page":"113","DOI":"10.1109\/TASSP.1979.1163209","article-title":"Suppression of acoustic noise in speech using spectral subtraction","volume":"27","author":"Boll","year":"1979","journal-title":"IEEE Trans. Acoust. Speech Signal Process."},{"key":"10.1016\/j.csl.2026.102023_b7","first-page":"436","article-title":"Nontexture inpainting by curvature-driven diffusions","volume":"12","author":"Chan","year":"2001"},{"issue":"5","key":"10.1016\/j.csl.2026.102023_b8","doi-asserted-by":"crossref","first-page":"466","DOI":"10.1109\/TSA.2003.811544","article-title":"Noise spectrum estimation in adverse environments: Improved minima controlled recursive averaging","volume":"11","author":"Cohen","year":"2003","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"10.1016\/j.csl.2026.102023_b9","first-page":"1200","article-title":"Region filling and object removal by exemplar-based image inpainting","volume":"13","author":"Criminisi","year":"2004"},{"issue":"4","key":"10.1016\/j.csl.2026.102023_b10","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1145\/3197517.3201357","article-title":"Looking to listen at the cocktail party: A speaker-independent audio-visual model for speech separation","volume":"37","author":"Ephrat","year":"2018","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.csl.2026.102023_b11","doi-asserted-by":"crossref","unstructured":"Foroushi, Z., Dansereau, R.M., 2024. Dynamic audio-visual speech enhancement using recurrent variational autoencoders. In: International Workshop on Acoustic Signal Enhancement. IWAENC.","DOI":"10.1109\/IWAENC61483.2024.10693981"},{"key":"10.1016\/j.csl.2026.102023_b12","doi-asserted-by":"crossref","unstructured":"Foroushi, Z., Dansereau, R.M., 2025. TITR\u2013CNN: Temporal Inpainting with Trend\u2013Remainder Decomposition for Partially Occluded Lip\u2013Region Video Reconstruction. In: Proceedings of the International Conference on Smart Multimedia. ICSM, Paris, France, Accepted.","DOI":"10.1109\/ICSM64417.2025.11541568"},{"key":"10.1016\/j.csl.2026.102023_b13","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2025.101923","article-title":"Enhanced audio-visual speech enhancement with posterior sampling methods in recurrent variational autoencoders","volume":"99","author":"Foroushi","year":"2026","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.csl.2026.102023_b14","doi-asserted-by":"crossref","unstructured":"Foroushi, Z., Dansereau, R.M., 2026b. Silence-Aware AV-RVAE with MALA-based Posterior Sampling for Speech Enhancement. In: 2026 IEEE International Conference on Consumer Electronics. ICCE, Dubai, UAE, http:\/\/dx.doi.org\/10.1109\/ICCE67443.2026.11449839.","DOI":"10.1109\/ICCE67443.2026.11449839"},{"key":"10.1016\/j.csl.2026.102023_b15","series-title":"Interspeech","article-title":"Visual speech enhancement","author":"Gabbay","year":"2018"},{"key":"10.1016\/j.csl.2026.102023_b16","series-title":"Dynamical variational autoencoders: A comprehensive review","author":"Girin","year":"2021"},{"key":"10.1016\/j.csl.2026.102023_b17","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Audio-visual speech enhancement with a deep Kalman filter generative model","author":"Golmakani","year":"2023"},{"key":"10.1016\/j.csl.2026.102023_b18","unstructured":"Google, 0000. WebRTC Voice Activity Detector. https:\/\/webrtc.org\/."},{"key":"10.1016\/j.csl.2026.102023_b19","first-page":"1","article-title":"Temporally coherent completion of dynamic video","author":"Huang","year":"2016","journal-title":"ACM Trans. Graph."},{"key":"10.1016\/j.csl.2026.102023_b20","first-page":"1","article-title":"Globally and locally consistent image completion","volume":"36","author":"Iizuka","year":"2017"},{"issue":"10","key":"10.1016\/j.csl.2026.102023_b21","doi-asserted-by":"crossref","first-page":"2737","DOI":"10.5194\/gmd-16-2737-2023","article-title":"CLGAN: A generative adversarial network (GAN)-based video prediction model for precipitation nowcasting","volume":"16","author":"Ji","year":"2023","journal-title":"I Geosci. Model. Dev."},{"key":"10.1016\/j.csl.2026.102023_b22","first-page":"694","article-title":"Perceptual losses for real-time style transfer and super-resolution","author":"Johnson","year":"2016"},{"key":"10.1016\/j.csl.2026.102023_b23","doi-asserted-by":"crossref","first-page":"1117","DOI":"10.1007\/s10772-023-10073-6","article-title":"An optimized convolutional neural network for speech enhancement","volume":"26","author":"Karthik","year":"2023","journal-title":"Int. J. Speech Technol."},{"key":"10.1016\/j.csl.2026.102023_b24","article-title":"Auto-encoding variational Bayes","author":"Kingma","year":"2014","journal-title":"Int. Conf. Learn. Represent. (ICLR2014)"},{"key":"10.1016\/j.csl.2026.102023_b25","first-page":"369","article-title":"Super-convergence: Very fast training of neural networks using large learning rates","volume":"11006","author":"L. N. Smith","year":"2019"},{"key":"10.1016\/j.csl.2026.102023_b26","doi-asserted-by":"crossref","first-page":"131205","DOI":"10.1109\/ACCESS.2024.3458460","article-title":"CNN-based time series decomposition model for video prediction","volume":"12","author":"Lee","year":"2024","journal-title":"IEEE Access J."},{"key":"10.1016\/j.csl.2026.102023_b27","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"371","article-title":"A recurrent variational autoencoder for speech enhancement","author":"Leglaive","year":"2020"},{"key":"10.1016\/j.csl.2026.102023_b28","series-title":"2018 IEEE 28th Intern. Workshop on Machine Learning for Signal Processing","first-page":"1","article-title":"A variance modeling framework based on variational autoencoders for speech enhancement","author":"Leglaive","year":"2018"},{"issue":"12","key":"10.1016\/j.csl.2026.102023_b29","doi-asserted-by":"crossref","first-page":"1586","DOI":"10.1109\/PROC.1979.11540","article-title":"Enhancement and bandwidth compression of noisy speech","volume":"67","author":"Lim","year":"1979","journal-title":"Proc. IEEE"},{"key":"10.1016\/j.csl.2026.102023_b30","doi-asserted-by":"crossref","first-page":"496","DOI":"10.1111\/rssb.12482","article-title":"The barker proposal: combining robustness and efficiency in gradient-based MCMC","volume":"84","author":"Livingstone","year":"2022","journal-title":"J. R. Stat. Soc. Ser. B R. Stat. Soc."},{"key":"10.1016\/j.csl.2026.102023_b31","first-page":"1256","article-title":"Conv-TasNet: Surpassing ideal time-frequency magnitude masking for speech separation","volume":"vol. 27, no. 8","author":"Luo","year":"2019"},{"issue":"5","key":"10.1016\/j.csl.2026.102023_b32","doi-asserted-by":"crossref","first-page":"504","DOI":"10.1109\/89.928915","article-title":"Noise power spectral density estimation based on optimal smoothing and minimum statistics","volume":"9","author":"Martin","year":"2001","journal-title":"IEEE Trans. Speech Audio Process."},{"key":"10.1016\/j.csl.2026.102023_b33","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"6319","article-title":"Lipreading using temporal convolutional networks","author":"Martinez","year":"2020"},{"key":"10.1016\/j.csl.2026.102023_b34","doi-asserted-by":"crossref","first-page":"1368","DOI":"10.1109\/TASLP.2021.3066303","article-title":"An overview of deep-learning-based audio-visual speech enhancement and separation","volume":"29","author":"Michelsanti","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"10","key":"10.1016\/j.csl.2026.102023_b35","doi-asserted-by":"crossref","first-page":"2140","DOI":"10.1109\/TASL.2013.2270369","article-title":"Supervised and unsupervised speech enhancement using nonnegative matrix factorization","volume":"21","author":"Mohammadiha","year":"2013","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102023_b36","series-title":"Handbook of Markov Chain Monte Carlo","article-title":"MCMC using Hamiltonian dynamics","author":"Neal","year":"2012"},{"key":"10.1016\/j.csl.2026.102023_b37","doi-asserted-by":"crossref","first-page":"1993","DOI":"10.1137\/140954933","article-title":"Video inpainting of complex scenes","author":"Newson","year":"2014","journal-title":"SIAM J. Imaging Sci. Soc. Ind. Appl. Math."},{"key":"10.1016\/j.csl.2026.102023_b38","doi-asserted-by":"crossref","DOI":"10.1109\/TASLP.2023.3285241","article-title":"Speech enhancement and dereverberation with diffusion-based generative models","author":"Richter","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102023_b39","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"749","article-title":"Perceptual evaluation of speech quality (PESQ)-a new method for speech quality assessment of telephone networks and codecs","author":"Rix","year":"2001"},{"key":"10.1016\/j.csl.2026.102023_b40","doi-asserted-by":"crossref","first-page":"337","DOI":"10.1023\/A:1023562417138","article-title":"Langevin diffusions and Metropolis-Hastings algorithms","volume":"4","author":"Roberts","year":"2002","journal-title":"Methodol. Comput. Appl. Probab."},{"key":"10.1016\/j.csl.2026.102023_b41","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"626","article-title":"SDR\u2013half-baked or well done?","author":"Roux","year":"2019"},{"key":"10.1016\/j.csl.2026.102023_b42","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"7534","article-title":"Robust unsupervised audio-visual speech enhancement using a mixture of variational autoencoders","author":"Sadeghi","year":"2020"},{"key":"10.1016\/j.csl.2026.102023_b43","doi-asserted-by":"crossref","first-page":"1788","DOI":"10.1109\/TASLP.2020.3000593","article-title":"Audio-visual speech enhancement using conditional variational auto-encoders","volume":"28","author":"Sadeghi","year":"2020","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102023_b44","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","article-title":"Posterior sampling algorithms for unsupervised speech enhancement with recurrent variational autoencoder","author":"Sadeghi","year":"2024"},{"key":"10.1016\/j.csl.2026.102023_b45","series-title":"Unsupervised learning of video representations using LSTMs","first-page":"843","author":"Srivastava","year":"2015"},{"issue":"7","key":"10.1016\/j.csl.2026.102023_b46","doi-asserted-by":"crossref","first-page":"2125","DOI":"10.1109\/TASL.2011.2114881","article-title":"An algorithm for intelligibility prediction of time-frequency weighted noisy speech","volume":"19","author":"Taal","year":"2011","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102023_b47","doi-asserted-by":"crossref","DOI":"10.52202\/075280-3060","article-title":"OpenSTL: A comprehensive benchmark of spatio-temporal predictive learning","author":"Tan","year":"2023"},{"key":"10.1016\/j.csl.2026.102023_b48","doi-asserted-by":"crossref","first-page":"205","DOI":"10.1016\/j.specom.2012.08.005","article-title":"Speech enhancement using hidden Markov models in Mel-frequency domain","volume":"55","author":"Veisi","year":"2013","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.102023_b49","series-title":"Audio Source Separation and Speech Enhancement","author":"Vincent","year":"2018"},{"key":"10.1016\/j.csl.2026.102023_b50","first-page":"600","article-title":"Image quality assessment: from error visibility to structural similarity","volume":"13","author":"Wang","year":"2004"},{"issue":"10","key":"10.1016\/j.csl.2026.102023_b51","doi-asserted-by":"crossref","first-page":"1702","DOI":"10.1109\/TASLP.2018.2842159","article-title":"Supervised speech separation based on deep learning: An overview","volume":"26","author":"Wang","year":"2018","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102023_b52","doi-asserted-by":"crossref","first-page":"3602","DOI":"10.1109\/TASLP.2023.3304482","article-title":"TF-GridNet: Integrating full- and sub-band modeling for speech separation","volume":"31","author":"Wang","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"12","key":"10.1016\/j.csl.2026.102023_b53","doi-asserted-by":"crossref","first-page":"1849","DOI":"10.1109\/TASLP.2014.2352935","article-title":"On training targets for supervised speech separation","volume":"22","author":"Wang","year":"2014","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"2","key":"10.1016\/j.csl.2026.102023_b54","first-page":"2208","article-title":"PredRNN: A recurrent neural network for spatiotemporal predictive learning","volume":"45","author":"Wang","year":"2023","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102023_b55","series-title":"Interspeech","first-page":"2928","article-title":"Speech enhancement with score-based generative models in the complex STFT domain","author":"Welker","year":"2022"},{"key":"10.1016\/j.csl.2026.102023_b56","first-page":"483","article-title":"Complex ratio masking for monaural speech separation","volume":"vol. 24, no. 3","author":"Williamson","year":"2016"},{"key":"10.1016\/j.csl.2026.102023_b57","series-title":"Webrtc VAD python interface","author":"Wiseman","year":"2016"},{"key":"10.1016\/j.csl.2026.102023_b58","first-page":"65","article-title":"An experimental study on speech enhancement based on deep neural networks","volume":"vol. 21, no. 1","author":"Xu","year":"2014"},{"issue":"1","key":"10.1016\/j.csl.2026.102023_b59","doi-asserted-by":"crossref","first-page":"7","DOI":"10.1109\/TASLP.2014.2364452","article-title":"A regression approach to speech enhancement based on deep neural networks","volume":"23","author":"Xu","year":"2015","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.102023_b60","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"1","article-title":"Cold diffusion for speech enhancement","author":"Yen","year":"2023"},{"key":"10.1016\/j.csl.2026.102023_b61","article-title":"Efficient and informationpreserving future frame prediction and beyond","author":"Yu","year":"2020","journal-title":"Int. Conf. Learn. Represent. (ICLR)"},{"key":"10.1016\/j.csl.2026.102023_b62","series-title":"AAAI","article-title":"Are transformers effective for time series forecasting?","author":"Zeng","year":"2023"},{"key":"10.1016\/j.csl.2026.102023_b63","doi-asserted-by":"crossref","first-page":"572","DOI":"10.1109\/TIP.2020.3036749","article-title":"A spatial-temporal recurrent neural network for video saliency prediction","volume":"30","author":"Zhang","year":"2020","journal-title":"IEEE Trans. Image Process."}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000860?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000860?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,7,11]],"date-time":"2026-07-11T19:36:24Z","timestamp":1783798584000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000860"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2027,2]]},"references-count":63,"alternative-id":["S0885230826000860"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.102023","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2027,2]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Robust audio\u2013visual speech enhancement under visual occlusion using TITR-CNN and silence-aware MALA-EM","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.102023","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"102023"}}