{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,17]],"date-time":"2026-02-17T12:12:23Z","timestamp":1771330343554,"version":"3.50.1"},"reference-count":29,"publisher":"IEEE","license":[{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2024,12,2]],"date-time":"2024-12-02T00:00:00Z","timestamp":1733097600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2024,12,2]]},"DOI":"10.1109\/slt61566.2024.10832325","type":"proceedings-article","created":{"date-parts":[[2025,1,16]],"date-time":"2025-01-16T18:31:27Z","timestamp":1737052287000},"page":"1131-1136","source":"Crossref","is-referenced-by-count":2,"title":["Self-Supervised Syllable Discovery Based on Speaker-Disentangled Hubert"],"prefix":"10.1109","author":[{"given":"Ryota","family":"Komatsu","sequence":"first","affiliation":[{"name":"Independent Researcher"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Takahiro","family":"Shinozaki","sequence":"additional","affiliation":[{"name":"Tokyo Institute of Technology"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-2044"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10446062"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10652"},{"key":"ref5","first-page":"1336","article-title":"On generative spoken language modeling from raw audio","volume":"9","author":"Lakhotia","year":"2021","journal-title":"TACL"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-475"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2023.3288409"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2023.findings-emnlp.1055"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-1718"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSPW62465.2024.10625802"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2022.naacl-main.63"},{"key":"ref12","article-title":"Transpeech: Speech-to-speech translation with bilateral perturbation","author":"Huang","year":"2023","journal-title":"ICLR"},{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP49357.2023.10096250"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2023-847"},{"key":"ref15","first-page":"4171","article-title":"Bert: Pretraining of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019","journal-title":"NAACL"},{"key":"ref16","first-page":"18003","article-title":"Contentvec: An improved self-supervised speech representation by disentangling speakers","volume":"162","author":"Qian","year":"2022","journal-title":"ICML"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1706.03762"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.3115\/1220175.1220179"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.5555\/3495724.3497510"},{"key":"ref22","first-page":"16251","article-title":"Neural analysis and synthesis: Reconstructing speech from selfsupervised representations","volume":"34","author":"Choi","year":"2021","journal-title":"NeurIPS"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2009-538"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/interspeech.2017-950"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1502.03167"},{"key":"ref26","article-title":"Gaussian error linear units (gelus)","author":"Hendrycks","year":"2016","journal-title":"arXiv:1606.08415"},{"key":"ref27","article-title":"Decoupled weight decay regularization","author":"Loshchilov","year":"2019","journal-title":"ICLR"},{"key":"ref28","article-title":"Dino as a von mises-fisher mixture model","author":"Govindarajan","year":"2023","journal-title":"ICLR"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1162\/tacl_a_00656"}],"event":{"name":"2024 IEEE Spoken Language Technology Workshop (SLT)","location":"Macao","start":{"date-parts":[[2024,12,2]]},"end":{"date-parts":[[2024,12,5]]}},"container-title":["2024 IEEE Spoken Language Technology Workshop (SLT)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/10830790\/10830793\/10832325.pdf?arnumber=10832325","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,1,20]],"date-time":"2025-01-20T18:39:17Z","timestamp":1737398357000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10832325\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,12,2]]},"references-count":29,"URL":"https:\/\/doi.org\/10.1109\/slt61566.2024.10832325","relation":{},"subject":[],"published":{"date-parts":[[2024,12,2]]}}}