{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T02:28:12Z","timestamp":1773887292037,"version":"3.50.1"},"reference-count":30,"publisher":"IEEE","license":[{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2020,5,1]],"date-time":"2020-05-01T00:00:00Z","timestamp":1588291200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2020,5]]},"DOI":"10.1109\/icassp40776.2020.9052940","type":"proceedings-article","created":{"date-parts":[[2020,4,9]],"date-time":"2020-04-09T16:21:13Z","timestamp":1586449273000},"page":"7054-7058","source":"Crossref","is-referenced-by-count":11,"title":["Sequence-Level Consistency Training for Semi-Supervised End-to-End Automatic Speech Recognition"],"prefix":"10.1109","author":[{"given":"Ryo","family":"Masumura","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mana","family":"Ihori","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Akihiko","family":"Takashima","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takafumi","family":"Moriya","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Atsushi","family":"Ando","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yusuke","family":"Shinohara","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref30","first-page":"947","article-title":"Spontaneous speech corpus of Japanese","author":"maekawa","year":"2000","journal-title":"Proc Int Conference on Language Resources and Evaluation (LREC)"},{"key":"ref10","first-page":"5884","article-title":"Speech-Transformer: A norecurrence sequence-to-sequence model for speech recognition","author":"dong","year":"2018","journal-title":"Proc International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref11","first-page":"7095","article-title":"The SpeechTrans- former for large-scale mandarin chinese speech recognition","author":"zhao","year":"2019","journal-title":"Proc International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2112"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1938"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462105"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268950"},{"key":"ref17","first-page":"477","article-title":"Leveraging sequence-to-sequence speech synthesis for enhancing acoustic-to-word speech recognition","author":"mimura","year":"2018","journal-title":"Proc IEEE\/ACL Workshop Spoken Lang Technol (SLT)"},{"key":"ref18","first-page":"426","article-title":"Back-translation-style data augmentation for end-to-end ASR","author":"hayashi","year":"2018","journal-title":"Proc IEEE\/ACL Workshop Spoken Lang Technol (SLT)"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1746"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683171"},{"key":"ref4","first-page":"193","article-title":"Exploring architectures, data and units for streaming end-to-end speech recognition with RNN-transducer","author":"rao","year":"2017","journal-title":"Proc Automatic Speech Recognition and Understanding Workshop (ASRU)"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1589"},{"key":"ref3","first-page":"1298","article-title":"Recurrent neural aligner: An encoder-decoder neural network model for sequence to sequence mapping","author":"sak","year":"2017","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref6","first-page":"3249","article-title":"A study of the recurrent neural network encoder-decoder for large vocabulary speech recognition","author":"lu","year":"2015","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2544"},{"key":"ref5","first-page":"4945","article-title":"End-to-end attention-based large vocabulary speech recognition","author":"bahdanau","year":"2015","journal-title":"Proc International Conference on Acoustics Speech and Signal Processing (ICASSP)"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472641"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472621"},{"key":"ref2","first-page":"959","article-title":"Direct acoustics-to-word models for English conversational speech recognition","author":"audhkhasi","year":"2017","journal-title":"Proc Annual Conference of the International Speech Communication Association (INTERSPEECH)"},{"key":"ref9","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"2017","journal-title":"Proc Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953069"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2019.8683307"},{"key":"ref22","article-title":"Temporal ensembling for semi- supervised learning","author":"laine","year":"2017","journal-title":"Proc of the Int Conf on Learning Representations (ICLR)"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-3167"},{"key":"ref24","article-title":"Virtual adversarial training: A regularization method for supervised and semi-supervised learning","author":"miyato","year":"2018","journal-title":"IEEE Transactions on Pattern Analysis and Machine Intelligence"},{"key":"ref23","first-page":"1195","article-title":"Weight-averaged consistency targets improve semi-supervised deep learning results","author":"tarvainen","year":"2017","journal-title":"Proc Advances in Neural Information Processing Systems (NIPS)"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D16-1139"},{"key":"ref25","article-title":"Unsupervised data augmentation for consistency training","author":"xie","year":"2019"}],"event":{"name":"ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Barcelona, Spain","start":{"date-parts":[[2020,5,4]]},"end":{"date-parts":[[2020,5,8]]}},"container-title":["ICASSP 2020 - 2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9040208\/9052899\/09052940.pdf?arnumber=9052940","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,6,27]],"date-time":"2022-06-27T20:24:35Z","timestamp":1656361475000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9052940\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2020,5]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/icassp40776.2020.9052940","relation":{},"subject":[],"published":{"date-parts":[[2020,5]]}}}