{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,22]],"date-time":"2026-04-22T19:44:15Z","timestamp":1776887055992,"version":"3.51.2"},"reference-count":89,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/creativecommons.org\/licenses\/by\/4.0\/legalcode"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Top. Signal Process."],"published-print":{"date-parts":[[2022,10]]},"DOI":"10.1109\/jstsp.2022.3195367","type":"journal-article","created":{"date-parts":[[2022,8,5]],"date-time":"2022-08-05T00:32:05Z","timestamp":1659659525000},"page":"1424-1438","source":"Crossref","is-referenced-by-count":20,"title":["Momentum Pseudo-Labeling: Semi-Supervised ASR With Continuously Improving Pseudo-Labels"],"prefix":"10.1109","volume":"16","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-4500-8957","authenticated-orcid":false,"given":"Yosuke","family":"Higuchi","sequence":"first","affiliation":[{"name":"Waseda University, Tokyo, Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Niko","family":"Moritz","sequence":"additional","affiliation":[{"name":"Mitsubishi Electric Research Laboratories (MERL), Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3451-171X","authenticated-orcid":false,"given":"Jonathan","family":"Le Roux","sequence":"additional","affiliation":[{"name":"Mitsubishi Electric Research Laboratories (MERL), Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-4560-8039","authenticated-orcid":false,"given":"Takaaki","family":"Hori","sequence":"additional","affiliation":[{"name":"Mitsubishi Electric Research Laboratories (MERL), Cambridge, MA, USA"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"ref3","first-page":"1764","article-title":"Towards end-to-end speech recognition with recurrent neural networks","volume-title":"Proc. 31st Int. Conf. Mach. Learn.","author":"Graves","year":"2014"},{"key":"ref4","first-page":"577","article-title":"Attention-based models for speech recognition","volume-title":"Proc. 28th Int. Conf. Neural Inf. Process. Syst.","author":"Chorowski","year":"2015"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-642-24797-2"},{"key":"ref8","first-page":"3104","article-title":"Sequence to sequence learning with neural networks","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Sutskever","year":"2014"},{"key":"ref9","article-title":"Neural machine translation by jointly learning to align and translate","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Bahdanau","year":"2014"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053889"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462105"},{"key":"ref14","first-page":"231","article-title":"RWTH ASR systems for LibriSpeech: Hybrid vs attention","volume-title":"Proc. Annu. Conf. Int. Speech Commun. Assoc.","author":"Lscher","year":"2019"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003750"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052942"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1007\/springerreference_63705"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268950"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683307"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413375"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1746"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639541"},{"key":"ref23","first-page":"5410","article-title":"Almost unsupervised text to speech and automatic speech recognition","volume-title":"Proc. 36th Int. Conf. Mach. Learn.","author":"Ren","year":"2019"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054224"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053176"},{"key":"ref27","first-page":"1597","article-title":"A simple framework for contrastive learning of visual representations","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Chen","year":"2020"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054438"},{"key":"ref30","article-title":"vq-wav2vec: Self-supervised learning of discrete speech representations","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Baevski","year":"2019"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414460"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1965.1053799"},{"key":"ref33","article-title":"Pseudo-label: The simple and efficient semi-supervised learning method for deep neural networks","volume-title":"Proc. Workshop Challenges Representation Learn.,","volume":"3","author":"Lee","year":"2013"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682172"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054295"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052940"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1337"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383552"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1800"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1280"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1470"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-740"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414299"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414058"},{"key":"ref45","article-title":"Temporal ensembling for semi-supervised learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Laine","year":"2017"},{"key":"ref46","first-page":"1195","article-title":"Mean teachers are better role models: Weight-averaged consistency targets improve semi-supervised deep learning results","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Tarvainen","year":"2017"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-571"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746275"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.00975"},{"key":"ref50","first-page":"21271","article-title":"Bootstrap your own latent - A new approach to self-supervised learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Grill","year":"2020"},{"key":"ref51","article-title":"Distilling the knowledge in a neural network","author":"Hinton","year":"2015"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461995"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1952"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3071662"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref56","article-title":"Unsupervised data augmentation for consistency training","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst.","author":"Xie","year":"2020"},{"key":"ref57","article-title":"Revisiting self-training for neural sequence generation","volume-title":"Proc. Int. Conf. Learn. Representations","author":"He","year":"2019"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR42600.2020.01070"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688028"},{"key":"ref60","first-page":"5998","article-title":"Attention is all you need","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Vaswani","year":"2017"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1296"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414858"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2017.195"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P19-1285"},{"key":"ref65","article-title":"Understanding and improving transformer from a multi-particle dynamic system point of view","author":"Lu","year":"2019"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2017-343"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688157"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413899"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1909"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2022-581"},{"key":"ref71","first-page":"448","article-title":"Batch normalization: Accelerating deep network training by reducing internal covariate shift","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Ioffe","year":"2015"},{"key":"ref72","first-page":"1942","article-title":"Batch renormalization: Towards reducing minibatch dependence in batch-normalized models","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Ioffe","year":"2017"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-030-01261-8_1"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-99579-3_21"},{"key":"ref76","first-page":"1","article-title":"The kaldi speech recognition toolkit","volume-title":"Proc. IEEE Autom. Speech Recognit. Understanding Workshop","author":"Povey","year":"2011"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P18-1007"},{"key":"ref78","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1456"},{"key":"ref79","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kingma","year":"2015"},{"key":"ref80","article-title":"First-pass large vocabulary continuous speech recognition using bi-directional recurrent DNNs","author":"Hannun","year":"2014"},{"key":"ref81","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003920"},{"key":"ref82","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2008-122"},{"key":"ref83","article-title":"Pushing the limits of semi-supervised learning for automatic speech recognition","author":"Zhang","year":"2020"},{"key":"ref84","article-title":"Instance normalization: The missing ingredient for fast stylization","author":"Ulyanov","year":"2016"},{"key":"ref85","article-title":"Layer normalization","author":"Ba","year":"2016"},{"key":"ref86","article-title":"Transformer versus LSTM language models trained on uncertain ASR hypotheses in limited data scenarios","author":"Sheikh","year":"2021"},{"key":"ref87","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-711"},{"key":"ref88","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-1488"},{"key":"ref89","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Baevski","year":"2020"}],"container-title":["IEEE Journal of Selected Topics in Signal Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/4200690\/9923627\/09846970.pdf?arnumber=9846970","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T09:26:12Z","timestamp":1706779572000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9846970\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10]]},"references-count":89,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/jstsp.2022.3195367","relation":{},"ISSN":["1932-4553","1941-0484"],"issn-type":[{"value":"1932-4553","type":"print"},{"value":"1941-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,10]]}}}