{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,6]],"date-time":"2026-05-06T15:07:48Z","timestamp":1778080068929,"version":"3.51.4"},"reference-count":33,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,6,4]]},"DOI":"10.1109\/icassp49357.2023.10095268","type":"proceedings-article","created":{"date-parts":[[2023,5,5]],"date-time":"2023-05-05T17:28:30Z","timestamp":1683307710000},"page":"1-5","source":"Crossref","is-referenced-by-count":4,"title":["Predicting Multi-Codebook Vector Quantization Indexes for Knowledge Distillation"],"prefix":"10.1109","author":[{"given":"Liyong","family":"Guo","sequence":"first","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Xiaoyu","family":"Yang","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Quandong","family":"Wang","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuxiang","family":"Kong","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zengwei","family":"Yao","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fan","family":"Cui","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Fangjun","family":"Kuang","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wei","family":"Kang","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Long","family":"Lin","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Mingshuang","family":"Luo","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Piotr","family":"\u017belasko","sequence":"additional","affiliation":[{"name":"Meaning.Team Inc,USA"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Daniel","family":"Povey","sequence":"additional","affiliation":[{"name":"Xiaomi Corp.,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413692"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2442"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8462619"},{"key":"ref14","article-title":"Deep learning of representations for unsupervised and transfer learning","author":"bengio","year":"2012","journal-title":"Proceedings of ICML Workshop on Unsupervised and Transfer Learning"},{"key":"ref31","article-title":"Unified streaming and non-streaming two-pass end-to-end model for speech recognition","author":"zhang","year":"2020"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-3015"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639635"},{"key":"ref33","doi-asserted-by":"crossref","DOI":"10.21437\/Interspeech.2017-1386","article-title":"Montreal forced aligner: Trainable text-speech alignment using kaldi","author":"mcauliffe","year":"2017","journal-title":"Proc INTERSPEECH"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003776"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10340"},{"key":"ref2","article-title":"Pushing the limits of semi-supervised learning for automatic speech recognition","author":"zhang","year":"2020","journal-title":"NeurIPS SAS Workshop"},{"key":"ref1","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","author":"baevski","year":"2020","journal-title":"NeurIPS Vancouver"},{"key":"ref17","article-title":"Beit: Bert pre-training of image transformers","author":"bao","year":"2021"},{"key":"ref16","article-title":"vq-wav2vec: Self-supervised learning of discrete speech representations","author":"baevski","year":"2019"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/83.480761"},{"key":"ref18","article-title":"Self-supervised learning with random-projection quantizer for speech recognition","author":"chiu","year":"2022"},{"key":"ref24","article-title":"Adam: A method for stochastic optimization","author":"kingma","year":"2014","journal-title":"Computer Science"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-797"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413905"},{"key":"ref25","author":"thomas","year":"2006","journal-title":"Elements of Information Theory"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/18.212286"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746168"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/DCC.1995.515515"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2680"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref29","article-title":"MU-SAN: A Music, Speech, and Noise Corpus","author":"snyder","year":"2015"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1873"},{"key":"ref7","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"devlin","year":"2019","journal-title":"NAACL"},{"key":"ref9","article-title":"Distilling the knowledge in a neural network","author":"hinton","year":"2014","journal-title":"NIPS Deep Learning Workshop"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-703"},{"key":"ref5","article-title":"A fine-tuned wav2vec 2.0\/hubert benchmark for speech emotion recognition, speaker verification and spoken language understanding","author":"wang","year":"2021"}],"event":{"name":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Rhodes Island, Greece","start":{"date-parts":[[2023,6,4]]},"end":{"date-parts":[[2023,6,10]]}},"container-title":["ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10094559\/10094560\/10095268.pdf?arnumber=10095268","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,27]],"date-time":"2023-11-27T18:59:21Z","timestamp":1701111561000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10095268\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,4]]},"references-count":33,"URL":"https:\/\/doi.org\/10.1109\/icassp49357.2023.10095268","relation":{},"subject":[],"published":{"date-parts":[[2023,6,4]]}}}