{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,21]],"date-time":"2025-05-21T14:26:30Z","timestamp":1747837590252,"version":"3.28.0"},"reference-count":29,"publisher":"IEEE","license":[{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,5,1]],"date-time":"2019-05-01T00:00:00Z","timestamp":1556668800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2019,5]]},"DOI":"10.1109\/icassp.2019.8682890","type":"proceedings-article","created":{"date-parts":[[2019,4,17]],"date-time":"2019-04-17T20:01:56Z","timestamp":1555531316000},"page":"6166-6170","source":"Crossref","is-referenced-by-count":31,"title":["Semi-supervised End-to-end Speech Recognition Using Text-to-speech and Autoencoders"],"prefix":"10.1109","author":[{"given":"Shigeki","family":"Karita","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tomoharu","family":"Iwata","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Marc","family":"Delcroix","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Atsunori","family":"Ogawa","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Tomohiro","family":"Nakatani","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1746"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2013.6707741"},{"key":"ref12","doi-asserted-by":"crossref","first-page":"949","DOI":"10.21437\/Interspeech.2017-1296","article-title":"Advances in Joint CTC-Attention Based End-to-End Speech Recognition with a Deep CNN Encoder and RNN-LM","author":"hori","year":"2017","journal-title":"InterSpeech"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268950"},{"key":"ref14","article-title":"Back-Translation-Style Data Augmentation for End-to-End ASR","author":"hayashi","year":"2018","journal-title":"proc of IEEE workshop on Spoken Language Technology (to appear)"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1162\/neco.2006.18.7.1527"},{"key":"ref16","doi-asserted-by":"crossref","first-page":"1045","DOI":"10.21437\/Interspeech.2010-343","article-title":"Recurrent Neural Network based Language Model","author":"mikolov","year":"2010","journal-title":"InterSpeech"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2016.07.004"},{"key":"ref18","first-page":"700","article-title":"Unsupervised Image-to-Image Translation Networks","author":"liu","year":"2017","journal-title":"Neural Information Processing Systems"},{"key":"ref19","article-title":"Unsupervised Machine Translation Using Monolingual Corpora Only","author":"lample","year":"2018","journal-title":"International Conference on Learning Representations"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref28","first-page":"1","article-title":"Adam: A Method for Stochastic Optimization","author":"kingma","year":"2014","journal-title":"International Conference on Learning Representations"},{"key":"ref3","article-title":"Deep Voice 3: 2000-Speaker Neural Text-to-Speech","author":"ping","year":"2017","journal-title":"CoRR"},{"key":"ref27","article-title":"X-vectors: Robust DNN Embeddings for Speaker Recognition","author":"snyder","year":"2018","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref6","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"2015","journal-title":"International Conference on Learning Representations"},{"key":"ref5","first-page":"3104","article-title":"Sequence to Sequence Learning with Neural Networks","author":"sutskever","year":"2014","journal-title":"Neural Information Processing Systems"},{"key":"ref29","doi-asserted-by":"crossref","DOI":"10.1109\/SLT.2018.8639693","article-title":"End-to-end speech recognition with word-based rnn language models","author":"hori","year":"2018"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511816338"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/MSP.2012.2205597"},{"key":"ref2","first-page":"4960","article-title":"Listen, attend and spell: A neural network for large vocabulary conversational speech recognition","author":"chan","year":"2016","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref1","first-page":"4945","article-title":"End-to-End Attention-based Large Vocabulary Speech Recognition","author":"bahdanau","year":"2016","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref9","first-page":"173","article-title":"Deep Speech 2: End-to-End Speech Recognition in English and Mandarin","volume":"48","author":"amodei","year":"2016","journal-title":"International Conference on Machine Learning"},{"key":"ref20","doi-asserted-by":"crossref","DOI":"10.1109\/SLT.2018.8639575","article-title":"Learning noise-invariant representations for robust speech recognition","author":"liang","year":"2018"},{"key":"ref22","first-page":"723","article-title":"A Kernel Two-sample Test","volume":"13","author":"gretton","year":"2012","journal-title":"J Mach Learn Res"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1456"},{"key":"ref24","article-title":"Very Deep Convolutional Networks for Large-Scale Image Recognition","author":"simonyan","year":"2015","journal-title":"International Conference on Learning Representations"},{"key":"ref23","first-page":"5206","article-title":"Lib-riSpeech: An ASR corpus based on public domain audio books","author":"panayotov","year":"2015","journal-title":"IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref26","first-page":"4006","article-title":"Tacotron: Towards End-to-End Speech Synthesis","author":"wang","year":"2017","journal-title":"Proceedings of International Conference on Spoken Language Processing IN-TERSPEECH"},{"article-title":"Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis","year":"2018","author":"jia","key":"ref25"}],"event":{"name":"ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","start":{"date-parts":[[2019,5,12]]},"location":"Brighton, United Kingdom","end":{"date-parts":[[2019,5,17]]}},"container-title":["ICASSP 2019 - 2019 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/8671773\/8682151\/08682890.pdf?arnumber=8682890","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,9,16]],"date-time":"2022-09-16T05:33:37Z","timestamp":1663306417000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8682890\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,5]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/icassp.2019.8682890","relation":{},"subject":[],"published":{"date-parts":[[2019,5]]}}}