{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,10,31]],"date-time":"2025-10-31T08:00:17Z","timestamp":1761897617328,"version":"3.37.3"},"reference-count":33,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,7,18]],"date-time":"2021-07-18T00:00:00Z","timestamp":1626566400000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100012166","name":"National Key Research and Development Program of China","doi-asserted-by":"publisher","award":["2018AAA0100500"],"award-info":[{"award-number":["2018AAA0100500"]}],"id":[{"id":"10.13039\/501100012166","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001809","name":"NSFC","doi-asserted-by":"publisher","award":["61832008,61772413"],"award-info":[{"award-number":["61832008,61772413"]}],"id":[{"id":"10.13039\/501100001809","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,7,18]]},"DOI":"10.1109\/ijcnn52387.2021.9533328","type":"proceedings-article","created":{"date-parts":[[2021,9,20]],"date-time":"2021-09-20T21:27:41Z","timestamp":1632173261000},"page":"1-8","source":"Crossref","is-referenced-by-count":4,"title":["Audio DistilBERT: A Distilled Audio BERT for Speech Representation Learning"],"prefix":"10.1109","author":[{"given":"Fan","family":"Yu","sequence":"first","affiliation":[{"name":"Xi&#x0027;an Jiaotong University, School of Computer Science &#x0026; Technology,Xi&#x0027;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Jiawei","family":"Guo","sequence":"additional","affiliation":[{"name":"Xi&#x0027;an Jiaotong University, School of Software Engineering,Xi&#x0027;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Xi","sequence":"additional","affiliation":[{"name":"Xi&#x0027;an Jiaotong University, School of Computer Science &#x0026; Technology,Xi&#x0027;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhao","family":"Yang","sequence":"additional","affiliation":[{"name":"Xi&#x0027;an Jiaotong University, School of Computer Science &#x0026; Technology,Xi&#x0027;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Rui","family":"Jiang","sequence":"additional","affiliation":[{"name":"Xi&#x0027;an Jiaotong University, School of Computer Science &#x0026; Technology,Xi&#x0027;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chao","family":"Zhang","sequence":"additional","affiliation":[{"name":"Xi&#x0027;an Jiaotong University, School of Computer Science &#x0026; Technology,Xi&#x0027;an,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref33","article-title":"The kaldi speech recognition toolkit","author":"povey","year":"0","journal-title":"2011 IEEE Workshop on Automatic Speech Recognition &amp; Understanding"},{"key":"ref32","first-page":"5998","article-title":"Attention is all you need","author":"vaswani","year":"0","journal-title":"Proceedings of the 31st International Conference on Neural Information Processing Systems"},{"doi-asserted-by":"publisher","key":"ref31","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref30","article-title":"Roberta: A robustly optimized BERT pretraining approach","author":"liu","year":"2019","journal-title":"vol abs\/1907 11692"},{"key":"ref10","first-page":"6889","article-title":"Unsupervised pre-training of bidirectional speech encoders via masked reconstruction","author":"wang","year":"0","journal-title":"2020 IEEE International Conference on Acoustics Speech and Signal Processing"},{"key":"ref11","article-title":"Self-supervised audio representation learning for mobile devices","author":"tagliasacchi","year":"2019","journal-title":"vol abs\/1905 11796"},{"doi-asserted-by":"publisher","key":"ref12","DOI":"10.1109\/ICASSP40776.2020.9053569"},{"key":"ref13","article-title":"BERT: pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"0","journal-title":"Proceedings of the 2019 Conference of the North American Chapter of the Association for Computational Linguistics Human Language Technologies 2019"},{"doi-asserted-by":"publisher","key":"ref14","DOI":"10.24963\/ijcai.2020\/341"},{"doi-asserted-by":"publisher","key":"ref15","DOI":"10.18653\/v1\/2020.acl-main.537"},{"key":"ref16","article-title":"Distilbert, a distilled version of BERT: smaller, faster, cheaper and lighter","author":"sanh","year":"2019","journal-title":"vol abs\/1910 01108"},{"doi-asserted-by":"publisher","key":"ref17","DOI":"10.18653\/v1\/D19-1441"},{"key":"ref18","first-page":"4163","article-title":"Tinybert: Distilling BERT for natural language understanding","author":"jiao","year":"0","journal-title":"Proceedings of the 2020 Conference on Empirical Methods in Natural Language Processing Findings"},{"doi-asserted-by":"publisher","key":"ref19","DOI":"10.18653\/v1\/2020.acl-main.195"},{"key":"ref28","article-title":"Paying more attention to attention: Improving the performance of convolutional neural networks via attention transfer","author":"zagoruyko","year":"0","journal-title":"5th International Conference on Learning Representations"},{"key":"ref4","article-title":"Improving transformer-based speech recognition using unsupervised pre-training","author":"jiang","year":"2019","journal-title":"vol abs\/1910 09932"},{"key":"ref27","article-title":"Fitnets: Hints for thin deep nets","author":"romero","year":"0","journal-title":"3rd International Conference on Learning Representations"},{"key":"ref3","first-page":"7694","article-title":"Effectiveness of self-supervised pretraining for ASR","author":"baevski","year":"0","journal-title":"2020 IEEE International Conference on Acoustics Speech and Signal Processing"},{"year":"2020","author":"liu","journal-title":"Tera Self-supervised learning of transformer encoder representation for speech","key":"ref6"},{"doi-asserted-by":"publisher","key":"ref29","DOI":"10.1609\/aaai.v34i05.6229"},{"doi-asserted-by":"publisher","key":"ref5","DOI":"10.1109\/ICASSP40776.2020.9054458"},{"year":"2018","author":"van den oord","journal-title":"Representation learning with contrastive predictive coding","key":"ref8"},{"year":"2020","author":"chi","journal-title":"Audio ALBERT A lite BERT for self-supervised learning of audio representation","key":"ref7"},{"key":"ref2","article-title":"vq-wav2vec: Self-supervised learning of discrete speech representations","author":"baevski","year":"0","journal-title":"International Conference on Learning Representations"},{"doi-asserted-by":"publisher","key":"ref9","DOI":"10.21437\/Interspeech.2019-1473"},{"doi-asserted-by":"publisher","key":"ref1","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref20","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","author":"baevski","year":"2020","journal-title":"Advances in neural information processing systems"},{"key":"ref22","article-title":"Efficient estimation of word representations in vector space","author":"mikolov","year":"0","journal-title":"1st International Conference on Learning Representations ICLR 2013"},{"doi-asserted-by":"publisher","key":"ref21","DOI":"10.1109\/ICASSP.2015.7178964"},{"doi-asserted-by":"publisher","key":"ref24","DOI":"10.1609\/aaai.v34i04.5963"},{"key":"ref23","article-title":"Distilling the knowledge in a neural network","author":"hinton","year":"2015","journal-title":"vol abs\/1503 02531"},{"doi-asserted-by":"publisher","key":"ref26","DOI":"10.1109\/CVPR42600.2020.01329"},{"doi-asserted-by":"publisher","key":"ref25","DOI":"10.1109\/CVPR.2019.00409"}],"event":{"name":"2021 International Joint Conference on Neural Networks (IJCNN)","start":{"date-parts":[[2021,7,18]]},"location":"Shenzhen, China","end":{"date-parts":[[2021,7,22]]}},"container-title":["2021 International Joint Conference on Neural Networks (IJCNN)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9533266\/9533267\/09533328.pdf?arnumber=9533328","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,8,2]],"date-time":"2022-08-02T23:32:43Z","timestamp":1659483163000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9533328\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,7,18]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/ijcnn52387.2021.9533328","relation":{},"subject":[],"published":{"date-parts":[[2021,7,18]]}}}