{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T15:24:20Z","timestamp":1775229860892,"version":"3.50.1"},"reference-count":80,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","license":[{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,1,1]],"date-time":"2023-01-01T00:00:00Z","timestamp":1672531200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"name":"Anhui Center for Applied Mathematics"},{"name":"Iflytek Company Ltd."},{"name":"Strategic Priority Research Program of Chinese Academy of Sciences","award":["XDC08010100"],"award-info":[{"award-number":["XDC08010100"]}]},{"DOI":"10.13039\/501100001809","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62101523"],"award-info":[{"award-number":["62101523"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]},{"name":"Hefei Municipal Natural Science Foundation","award":["2022012"],"award-info":[{"award-number":["2022012"]}]},{"name":"USTC Research Funds of the Double First-Class Initiative","award":["YD2100002008"],"award-info":[{"award-number":["YD2100002008"]}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2023]]},"DOI":"10.1109\/taslp.2023.3313434","type":"journal-article","created":{"date-parts":[[2023,9,11]],"date-time":"2023-09-11T19:11:16Z","timestamp":1694459476000},"page":"3908-3921","source":"Crossref","is-referenced-by-count":13,"title":["A Semi-Supervised Complementary Joint Training Approach for Low-Resource Speech Recognition"],"prefix":"10.1109","volume":"31","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-6176-8676","authenticated-orcid":false,"given":"Ye-Qian","family":"Du","sequence":"first","affiliation":[{"name":"School of Data Science, University of Science and Technology of China, Hefei, Anhui, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1124-0854","authenticated-orcid":false,"given":"Jie","family":"Zhang","sequence":"additional","affiliation":[{"name":"Department of Electronic Engineering and Information Science, University of Science and Technology of China, Hefei, Anhui, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xin","family":"Fang","sequence":"additional","affiliation":[{"name":"iFlytek Research, Hefei, Anhui, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Ming-Hui","family":"Wu","sequence":"additional","affiliation":[{"name":"iFlytek Research, Hefei, Anhui, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9454-9146","authenticated-orcid":false,"given":"Zhou-Wang","family":"Yang","sequence":"additional","affiliation":[{"name":"School of Mathematical Sciences, University of Science and Technology of China, Hefei, Anhui, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref57","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2011-358"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3184480"},{"key":"ref56","first-page":"23","article-title":"The design for the wall street journal-based CSR corpus","author":"paul","year":"0","journal-title":"Proc Workshop Speech Natural Lang"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref59","first-page":"5732","article-title":"Insights on representational similarity in neural networks with canonical correlation","author":"morcos","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462331"},{"key":"ref58","first-page":"2613","article-title":"A complementary joint training approach using unpaired speech and text for low-resource automatic speech recognition","author":"du","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2016-1596"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414641"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688018"},{"key":"ref55","article-title":"Transformers with convolutional context for ASR","author":"mohamed","year":"2019"},{"key":"ref11","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","author":"baevski","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052942"},{"key":"ref10","article-title":"Representation learning with contrastive predictive coding","author":"van den oord","year":"2018"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1800"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054295"},{"key":"ref19","article-title":"slimIPL: Language-model-free iterative pseudo-labeling","author":"likhomanenko","year":"2020"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054309"},{"key":"ref51","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414080"},{"key":"ref50","article-title":"Unsupervised pre-training for sequence to sequence speech recognition","author":"fan","year":"2019"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1746"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639541"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.acl-long.393"},{"key":"ref47","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2456"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682890"},{"key":"ref41","first-page":"5410","article-title":"Almost unsupervised text to speech and automatic speech recognition","author":"ren","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413375"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3167"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/SLT54892.2023.10022448"},{"key":"ref8","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"kenton","year":"0","journal-title":"Proc Conf North Amer Chapter Assoc Comput Linguistics - Hum Lang Technol"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1473"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3095662"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462506"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1002\/j.1538-7305.1970.tb04297.x"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"ref40","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268950"},{"key":"ref80","first-page":"27 826","article-title":"Unsupervised speech recognition","author":"baevski","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688009"},{"key":"ref79","article-title":"Applying wav2vec2. 0 to speech recognition in various low-resource languages","author":"yi","year":"2020"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.findings-emnlp.236"},{"key":"ref78","article-title":"Multilingual end-to-end speech recognition with a single transformer on low-resource languages","author":"zhou","year":"2018"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053831"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639619"},{"key":"ref31","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383552"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1511"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1179"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1574"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1109\/LSP.2021.3071668"},{"key":"ref77","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-909"},{"key":"ref32","first-page":"7059","article-title":"Independent language modeling architecture for end-to-end ASR","author":"xu","year":"0","journal-title":"Proc IEEE Int Conf Acoust Speech Signal Process"},{"key":"ref76","article-title":"Data techniques for online end-to-end speech recognition","author":"chen","year":"2020"},{"key":"ref2","author":"graves","year":"2012","journal-title":"Sequence transduction with recurrent neural networks"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1882"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746217"},{"key":"ref71","article-title":"Effectiveness of self-supervised pre-training for speech recognition","author":"baevski","year":"2019"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-4009"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1280"},{"key":"ref72","article-title":"DeCoAR 2.0: Deep contextualized acoustic representations with vector quantization","author":"ling","year":"2020"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3246"},{"key":"ref68","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-2074"},{"key":"ref67","article-title":"MUSAN: A music, speech, and noise corpus","author":"snyder","year":"2015"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8682172"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-343"},{"key":"ref25","article-title":"Semi-supervised speech recognition via local prior matching","author":"hsu","year":"2020"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1337"},{"key":"ref63","first-page":"707","article-title":"Binary codes capable of correcting deletions, insertions, and reversals","volume":"10","author":"levenshtein","year":"0","journal-title":"Sov Phys Doklady"},{"key":"ref66","first-page":"17022","article-title":"HiFi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis","author":"kong","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1385"},{"key":"ref65","article-title":"FastSpeech 2: Fast and high-quality end-to-end text to speech","author":"ren","year":"2020"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746249"},{"key":"ref28","first-page":"1081","article-title":"Effective sentence scoring method using BERT for speech recognition","author":"shin","year":"0","journal-title":"Proc Asian Conf Mach Learn"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462682"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1554"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2013.6707741"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414299"},{"key":"ref61","first-page":"592","article-title":"High quality agreement-based semi-supervised training data for acoustic modeling","author":"quitry","year":"0","journal-title":"Proc IEEE Spoken Lang Technol Workshop"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/9970249\/10246368.pdf?arnumber=10246368","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,13]],"date-time":"2023-11-13T19:35:13Z","timestamp":1699904113000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10246368\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023]]},"references-count":80,"URL":"https:\/\/doi.org\/10.1109\/taslp.2023.3313434","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2023]]}}}