{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,4]],"date-time":"2026-07-04T16:38:02Z","timestamp":1783183082513,"version":"3.54.6"},"reference-count":80,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"12","license":[{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2019,12,1]],"date-time":"2019-12-01T00:00:00Z","timestamp":1575158400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE\/ACM Trans. Audio Speech Lang. Process."],"published-print":{"date-parts":[[2019,12]]},"DOI":"10.1109\/taslp.2019.2938863","type":"journal-article","created":{"date-parts":[[2019,9,3]],"date-time":"2019-09-03T00:35:26Z","timestamp":1567470926000},"page":"2041-2053","source":"Crossref","is-referenced-by-count":198,"title":["Unsupervised Speech Representation Learning Using WaveNet Autoencoders"],"prefix":"10.1109","volume":"27","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-1570-7610","authenticated-orcid":false,"given":"Jan","family":"Chorowski","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-2010-4053","authenticated-orcid":false,"given":"Ron J.","family":"Weiss","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Samy","family":"Bengio","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Aaron","family":"van den Oord","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2148"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1160"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2016.2530409"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2016.04.033"},{"key":"ref76","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-2341"},{"key":"ref77","article-title":"Unsupervised machine translation using monolingual corpora only","author":"lample","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.909282"},{"key":"ref39","article-title":"Beta-VAE: Learning basic visual concepts with a constrained variational framework","author":"higgins","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref75","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2017.04.008"},{"key":"ref38","article-title":"Auto-encoding variational Bayes","author":"kingma","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref78","first-page":"7365","article-title":"Unsupervised cross-modal alignment of speech and text embedding spaces","author":"chung","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2011.6163965"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D15-1075"},{"key":"ref32","first-page":"6294","article-title":"Learned in translation: Contextualized word vectors","author":"mccann","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref31","first-page":"237","article-title":"Improved bottleneck features using pretrained deep neural networks","author":"yu","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2012.6424246"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1038\/381607a0"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1038\/44565"},{"key":"ref35","article-title":"Continuous latent variables","author":"bishop","year":"2006","journal-title":"Pattern Recognition and Machine Learning"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/D17-1070"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8269013"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8269010"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1121\/1.422247"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2009.5206848"},{"key":"ref64","first-page":"1068","article-title":"Neural audio synthesis of musical notes with wavenet autoencoders","author":"engel","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref27","first-page":"1","article-title":"Learning the speech front-end with raw waveform CLDNNs","author":"sainath","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref65","first-page":"1","article-title":"Domain-adversarial training of neural networks","volume":"17","author":"ganin","year":"2016","journal-title":"J Mach Learn Res"},{"key":"ref66","first-page":"1876","article-title":"Unsupervised learning of disentangled and interpretable representations from sequential data","author":"hsu","year":"2017","journal-title":"Adv Neural Inf Process Syst"},{"key":"ref29","first-page":"3320","article-title":"How transferable are features in deep neural networks?","author":"yosinski","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref67","first-page":"5656","article-title":"Disentangled sequential autoencoder","author":"li","year":"0","journal-title":"Proc 35th Int Conf Mach Learn"},{"key":"ref68","first-page":"3189","article-title":"Parallel inference of Dirichlet process Gaussian mixture models for unsupervised acoustic modeling: A feasibility study","author":"chen","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref69","first-page":"40","article-title":"A nonparametric Bayesian approach to acoustic model discovery","author":"lee","year":"0","journal-title":"Proc Annual Meeting of the Assoc Computational Linguistics"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390294"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1126\/science.1127647"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8268953"},{"key":"ref22","first-page":"873","article-title":"Sparse deep belief net model for visual area V2","author":"lee","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1038\/323533a0"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2011.5947700"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6854950"},{"key":"ref26","first-page":"11","article-title":"Analysis of CNN-based speech recognition system using raw speech as input","author":"palaz","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref25","first-page":"890","article-title":"Acoustic modeling with deep neural networks using raw time signal for LVCSR","author":"t\u00fcske","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref50","first-page":"1929","article-title":"Dropout: A simple way to prevent neural networks from overfitting","volume":"15","author":"hinton","year":"2014","journal-title":"J Mach Learn Res"},{"key":"ref51","article-title":"Zoneout: Regularizing RNNs by randomly preserving hidden activations","author":"krueger","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref59","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2017.8269009"},{"key":"ref58","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2016.04.032"},{"key":"ref57","first-page":"915","article-title":"Evaluating speech features with the minimal-pair ABX task (II): Resistance to noise","author":"schatz","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref56","first-page":"1","article-title":"Evaluating speech features with the minimal-pair ABX task: Analysis of the classical MFC\/PLP pipeline","author":"schatz","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref55","article-title":"The Kaldi speech recognition toolkit","author":"povey","year":"0","journal-title":"Proc IEEE Workshop on Automatic Speech Recognition and Understanding"},{"key":"ref54","first-page":"9084","article-title":"Invariant representations without adversarial training","volume":"31","author":"moyer","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref53","doi-asserted-by":"publisher","DOI":"10.1137\/0330046"},{"key":"ref52","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"0","journal-title":"Proc 3rd Int Conf Learn Representations"},{"key":"ref10","article-title":"QANet: Combining local convolution with global self-attention for reading comprehension","author":"yu","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-10590-1_53"},{"key":"ref40","article-title":"Deep variational information bottleneck","author":"alemi","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref12","first-page":"3319","article-title":"Axiomatic attribution for deep networks","author":"sundararajan","year":"0","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref13","first-page":"2564","article-title":"Understanding the representation and computation of multilayer perceptrons: A case study in speech recognition","author":"nagamine","year":"0","journal-title":"Proc 34th Int Conf Mach Learn"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461282"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2012.6424230"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638959"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref18","article-title":"WaveNet: A generative model for raw audio","author":"van den oord","year":"2016"},{"key":"ref19","first-page":"6309","article-title":"Neural discrete representation learning","author":"van den oord","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref80","article-title":"The zero resource speech challenge 2019: TTS without T","author":"dunbar","year":"0","journal-title":"Proc INTERSPEECH"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2015.7298594"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1145\/3065386"},{"key":"ref6","article-title":"Google's neural machine translation system: Bridging the gap between human and machine translation","author":"wu","year":"2016"},{"key":"ref5","article-title":"Neural machine translation by jointly learning to align and translate","author":"bahdanau","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462105"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6638947"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/K16-1002"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/P17-1018"},{"key":"ref46","article-title":"Hierarchical generative modeling for controllable speech synthesis","author":"hsu","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref45","article-title":"An empirical evaluation of generic convolutional and recurrent networks for sequence modeling","author":"bai","year":"2018"},{"key":"ref48","doi-asserted-by":"publisher","DOI":"10.1162\/089976602317318938"},{"key":"ref47","article-title":"PixelVAE: A latent variable model for natural images","author":"gulrajani","year":"0","journal-title":"Proc Int Conf Learn Representations"},{"key":"ref42","article-title":"Estimating or propagating gradients through stochastic neurons for conditional computation","author":"bengio","year":"2013"},{"key":"ref41","first-page":"4743","article-title":"Improved variational inference with inverse autoregressive flow","author":"kingma","year":"0","journal-title":"Proc Adv Neural Inf Process Syst"},{"key":"ref44","first-page":"933","article-title":"Language modeling with gated convolutional networks","author":"dauphin","year":"0","journal-title":"Proc Int Conf Mach Learn"},{"key":"ref43","author":"jurafsky","year":"2009","journal-title":"Speech and Language Processing"}],"container-title":["IEEE\/ACM Transactions on Audio, Speech, and Language Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/6570655\/8817316\/08822475.pdf?arnumber=8822475","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,7,13]],"date-time":"2022-07-13T20:54:26Z","timestamp":1657745666000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/8822475\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2019,12]]},"references-count":80,"journal-issue":{"issue":"12"},"URL":"https:\/\/doi.org\/10.1109\/taslp.2019.2938863","relation":{},"ISSN":["2329-9290","2329-9304"],"issn-type":[{"value":"2329-9290","type":"print"},{"value":"2329-9304","type":"electronic"}],"subject":[],"published":{"date-parts":[[2019,12]]}}}