{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,1,18]],"date-time":"2025-01-18T05:06:58Z","timestamp":1737176818864,"version":"3.33.0"},"reference-count":47,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,2]]},"DOI":"10.1109\/slt61566.2024.10832141","type":"proceedings-article","created":{"date-parts":[[2025,1,16]],"date-time":"2025-01-16T18:31:27Z","timestamp":1737052287000},"page":"200-207","source":"Crossref","is-referenced-by-count":0,"title":["Efficient Extraction of Noise-Robust Discrete Units from Self-Supervised Speech Models"],"prefix":"10.1109","author":[{"given":"Jakob","family":"Poncelet","sequence":"first","affiliation":[{"name":"KU Leuven,ESAT-PSI,Department Electrical Engineering,Leuven,Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yujun","family":"Wang","sequence":"additional","affiliation":[{"name":"Xiaomi Corporation,Beijing,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hugo Van","family":"Hamme","sequence":"additional","affiliation":[{"name":"KU Leuven,ESAT-PSI,Department Electrical Engineering,Leuven,Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref3","first-page":"12449","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume-title":"Proc. NeurIPS","author":"Baevski"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3207050"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096149"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2021-1775"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-143"},{"article-title":"vq-wav2vec: Self-supervised learning of discrete speech representations","volume-title":"Proc. ICLR","author":"Baevski","key":"ref8"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688253"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447929"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096788"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.63"},{"key":"ref13","first-page":"1336","article-title":"On generative spoken language modeling from raw audio","volume":"9","author":"Lakhotia","year":"2021","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054224"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096988"},{"key":"ref16","first-page":"27826","article-title":"Unsupervised speech recognition","volume-title":"Proc. NeurIPS","author":"Baevski"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-236"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSPW59220.2023.10193184"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-2192"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446344"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096603"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/SLT54892.2023.10022474"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.iwslt-1.46"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.4324\/9781003022022-6"},{"article-title":"FATHuBERT: Front-end adaptive training of hidden-unit BERT for distortion-invariant robust speech recognition","volume-title":"Proc. ASRU","author":"Yang","key":"ref25"},{"key":"ref26","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3275033"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096308"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10095480"},{"article-title":"Parameter-efficient transfer learning for NLP","volume-title":"Proc. ICML","author":"Houlsby","key":"ref29"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-2051"},{"key":"ref31","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","volume-title":"Proc. NAACL","author":"Devlin"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00656"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/OJSP.2020.3045349"},{"article-title":"LoRA: Low-rank adaptation of large language models","volume-title":"Proc. ICLR","author":"Edward Hu","key":"ref35"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref37","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2018-1456"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953152"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1121\/1.4806631"},{"key":"ref40","article-title":"MUSAN: A music, speech, and noise corpus","author":"Snyder","year":"2015","journal-title":"arXiv:1510.08484"},{"key":"ref41","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2015.7404837"},{"key":"ref42","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-4009"},{"key":"ref43","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2017.2763455"},{"key":"ref44","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2020-3015"},{"key":"ref45","doi-asserted-by":"publisher","DOI":"10.1109\/SLT54892.2023.10022656"},{"key":"ref46","doi-asserted-by":"publisher","DOI":"10.5555\/3295222.3295349"},{"article-title":"Ambient noise database for telephonometry","year":"1996","author":"Advanced Technology (NTT-AT)","key":"ref47"}],"event":{"name":"2024 IEEE Spoken Language Technology Workshop (SLT)","start":{"date-parts":[[2024,12,2]]},"location":"Macao","end":{"date-parts":[[2024,12,5]]}},"container-title":["2024 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10830790\/10830793\/10832141.pdf?arnumber=10832141","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,17]],"date-time":"2025-01-17T07:49:51Z","timestamp":1737100191000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10832141\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"references-count":47,"URL":"https:\/\/doi.org\/10.1109\/slt61566.2024.10832141","relation":{},"subject":[],"published":{"date-parts":[[2024,12,2]]}}}