{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,7]],"date-time":"2026-07-07T06:26:13Z","timestamp":1783405573499,"version":"3.54.6"},"reference-count":79,"publisher":"Institute of Electrical and Electronics Engineers (IEEE)","issue":"6","license":[{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/ieeexplore.ieee.org\/Xplorehelp\/downloads\/license-information\/IEEE.html"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2022,10,1]],"date-time":"2022-10-01T00:00:00Z","timestamp":1664582400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":["IEEE J. Sel. Top. Signal Process."],"published-print":{"date-parts":[[2022,10]]},"DOI":"10.1109\/jstsp.2022.3203608","type":"journal-article","created":{"date-parts":[[2022,9,1]],"date-time":"2022-09-01T19:39:27Z","timestamp":1662061167000},"page":"1380-1390","source":"Crossref","is-referenced-by-count":17,"title":["Autoregressive Predictive Coding: A Comprehensive Study"],"prefix":"10.1109","volume":"16","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3786-3251","authenticated-orcid":false,"given":"Gene-Ping","family":"Yang","sequence":"first","affiliation":[{"name":"University of Edinburgh, Edinburgh, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sung-Lin","family":"Yeh","sequence":"additional","affiliation":[{"name":"University of Edinburgh, Edinburgh, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-9451-7956","authenticated-orcid":false,"given":"Yu-An","family":"Chung","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-3097-360X","authenticated-orcid":false,"given":"James","family":"Glass","sequence":"additional","affiliation":[{"name":"Massachusetts Institute of Technology, Cambridge, MA, USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-2445-2605","authenticated-orcid":false,"given":"Hao","family":"Tang","sequence":"additional","affiliation":[{"name":"University of Edinburgh, Edinburgh, U.K."}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1002\/j.1538-7305.1948.tb01338.x"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1955.1055126"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TIT.1955.1055116"},{"key":"ref4","article-title":"Relating data compression and learnability","author":"Littlestone","year":"1986"},{"key":"ref5","first-page":"2792","article-title":"On statistical learning via the lens of compression","volume-title":"Proc. 30th Int. Conf. Neural Inf. Process. Syst.","author":"David","year":"2016"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N18-1202"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1810.04805"},{"key":"ref8","first-page":"5753","article-title":"XLNet: Generalized autoregressive pretraining for language understanding","volume-title":"Proc. 33rd Int. Conf. Neural Inf. Process. Syst.","author":"Yang","year":"2019"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref10","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. 34th Int. Conf. Neural Inf. Process. Syst.","author":"Baevski","year":"2020"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053176"},{"key":"ref12","article-title":"DeCoAR 2.0: Deep contextualized acoustic representations with vector quantization","author":"Ling","year":"2021"},{"key":"ref13","article-title":"Improving transformer-based speech recognition using unsupervised pre-training","author":"Jiang","year":"2019","journal-title":"arXiv:1910.09932"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414539"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1473"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1109\/jstsp.2022.3182537"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2017-1452"},{"key":"ref18","article-title":"Representation learning with contrastive predictive coding","author":"van den Oord","year":"2018","journal-title":"arXiv:1807.03748"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-391"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3095662"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/tkde.2021.3090866"},{"key":"ref22","article-title":"Mixture density networks","author":"Bishop","year":"1994"},{"key":"ref23","first-page":"6309","article-title":"Neural discrete representation learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"van den Oord","year":"2017"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053541"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390294"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1228"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1544"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2022.3180684"},{"key":"ref29","article-title":"Information theoretic co-training","author":"McAllester","year":"2018","journal-title":"arXiv:1802.07572"},{"key":"ref30","article-title":"A mutual information maximization perspective of language representation learning","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kong","year":"2020"},{"key":"ref31","article-title":"Representation learning for sequence data with deep autoencoding predictive components","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Bai","year":"2021"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1007\/978-1-4615-3650-5"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.21437\/SSW.2016"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2019.2938863"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054438"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1145\/1390156.1390177"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.1145\/1553374.1553469"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/TPAMI.2013.50"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV.2015.167"},{"key":"ref40","first-page":"3079","article-title":"Semi-supervised sequence learning","volume-title":"Proc. 28th Int. Conf. Neural Inf. Process. Syst.","author":"Dai","year":"2015"},{"key":"ref41","first-page":"843","article-title":"Unsupervised learning of video representations using LSTMs","volume-title":"Proc. 32nd Int. Conf. Mach. Learn.","author":"Srivastava","year":"2015"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2014.81"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46493-0_35"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.1007\/978-3-319-46466-4_5"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2016.278"},{"key":"ref46","first-page":"2438","article-title":"Analyzing hidden representations in end-to-end automatic speech recognition systems","volume-title":"Proc. 31st Int. Conf. Neural Inf. Process. Syst.","author":"Belinkov","year":"2017"},{"key":"ref47","article-title":"What do end-to-end speech models learn about speaker, language and channel information? A layer-wise and neuron-level analysis","author":"Chowdhury","year":"2021","journal-title":"arXiv:2107.00439"},{"key":"ref48","article-title":"Pushing the limits of semi-supervised learning for automatic speech recognition","volume-title":"Proc. NeurIPS SAS Workshop","author":"Zhang","year":"2020"},{"key":"ref49","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688253"},{"key":"ref50","article-title":"UniSpeech at scale: An empirical study of pre-training method on large-scale speech recognition dataset","author":"Wang","year":"2021","journal-title":"arXiv:2101.07597"},{"key":"ref51","first-page":"10937","article-title":"UniSpeech: Unified speech representation learning with labeled and unlabeled data","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Wang","year":"2021"},{"key":"ref52","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-654"},{"key":"ref53","first-page":"1177","article-title":"Random features for large-scale kernel machines","volume-title":"Proc. 20th Int. Conf. Neural Inf. Process. Syst.","author":"Rahimi","year":"2007"},{"key":"ref54","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref55","doi-asserted-by":"publisher","DOI":"10.3115\/1075527.1075614"},{"key":"ref56","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2019.101027"},{"key":"ref57","article-title":"The Kaldi speech recognition toolkit","volume-title":"Proc. IEEE Workshop Autom. Speech Recognit. Understanding","author":"Povey","year":"2011"},{"key":"ref58","first-page":"249","article-title":"Understanding the difficulty of training deep feedforward neural networks","volume-title":"Proc. 13th Int. Conf. Artif. Intell. Statist.","author":"Glorot","year":"2010"},{"key":"ref59","article-title":"Adam: A method for stochastic optimization","volume-title":"Proc. Int. Conf. Learn. Representations","author":"Kingma","year":"2015"},{"key":"ref60","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2015-350"},{"key":"ref61","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639517"},{"key":"ref62","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-349"},{"key":"ref63","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2017.2752462"},{"key":"ref64","doi-asserted-by":"publisher","DOI":"10.1080\/23273798.2014.963130"},{"key":"ref65","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2014.6853678"},{"key":"ref66","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472655"},{"key":"ref67","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953075"},{"key":"ref68","first-page":"577","article-title":"Attention-based models for speech recognition","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Chorowski","year":"2015"},{"key":"ref69","doi-asserted-by":"publisher","DOI":"10.3115\/v1\/w14-4012"},{"key":"ref70","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1488"},{"key":"ref71","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2017-343"},{"key":"ref72","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639622"},{"key":"ref73","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-1280"},{"key":"ref74","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-620"},{"key":"ref75","first-page":"1214","article-title":"VAE with a VampPrior","volume-title":"Proc. 21st Int. Conf. Artif. Intell. Statist.","author":"Tomczak","year":"2018"},{"key":"ref76","first-page":"5580","article-title":"Multimodal generative models for scalable weakly-supervised learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Wu","year":"2018"},{"key":"ref77","first-page":"309","article-title":"Predicting what you already know helps: Provable self-supervised learning","volume-title":"Proc. Int. Conf. Neural Inf. Process. Syst.","author":"Lee","year":"2021"},{"key":"ref78","first-page":"5628","article-title":"A theoretical analysis of contrastive unsupervised representation learning","volume-title":"Proc. Int. Conf. Mach. Learn.","author":"Arora","year":"2019"},{"key":"ref79","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2020.acl-main.463"}],"container-title":["IEEE Journal of Selected Topics in Signal Processing"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/4200690\/9923627\/09874771.pdf?arnumber=9874771","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,2,1]],"date-time":"2024-02-01T12:33:58Z","timestamp":1706790838000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9874771\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2022,10]]},"references-count":79,"journal-issue":{"issue":"6"},"URL":"https:\/\/doi.org\/10.1109\/jstsp.2022.3203608","relation":{},"ISSN":["1932-4553","1941-0484"],"issn-type":[{"value":"1932-4553","type":"print"},{"value":"1941-0484","type":"electronic"}],"subject":[],"published":{"date-parts":[[2022,10]]}}}