{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,2]],"date-time":"2026-07-02T09:47:37Z","timestamp":1782985657193,"version":"3.54.5"},"reference-count":39,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,6,4]]},"DOI":"10.1109\/icassp49357.2023.10095280","type":"proceedings-article","created":{"date-parts":[[2023,5,5]],"date-time":"2023-05-05T17:28:30Z","timestamp":1683307710000},"page":"1-5","source":"Crossref","is-referenced-by-count":1,"title":["Fully Unsupervised Topic Clustering of Unlabelled Spoken Audio Using Self-Supervised Representation Learning and Topic Model"],"prefix":"10.1109","author":[{"given":"Takashi","family":"Maekaku","sequence":"first","affiliation":[{"name":"Yahoo Japan Corporation,Tokyo,Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Yuya","family":"Fujita","sequence":"additional","affiliation":[{"name":"Yahoo Japan Corporation,Tokyo,Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Xuankai","family":"Chang","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University,PA,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Shinji","family":"Watanabe","sequence":"additional","affiliation":[{"name":"Carnegie Mellon University,PA,USA"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref13","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2011-619"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1965"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2013.6639335"},{"key":"ref34","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052942"},{"key":"ref15","article-title":"HuBERT: How much can a bad teacher benefit ASR pre-training","author":"hsu","year":"2020","journal-title":"Workshop on Self-Supervised Learning for Speech and Audio Processing NeurIPS"},{"key":"ref37","article-title":"The fisher corpus: A resource for the next generations of speech-to-text","author":"cieri","year":"2004","journal-title":"Proceedings of the Fourth International Conference on Language Resources and Evaluation (LREC&#x2019;04)"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2007.909282"},{"key":"ref36","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2021.acl-long.80"},{"key":"ref31","article-title":"Wav2seq: Pre-training speech-to-text encoder-decoder models using pseudo languages","author":"wu","year":"2022"},{"key":"ref30","first-page":"23","article-title":"A new algorithm for data compression","volume":"12","author":"gage","year":"1994","journal-title":"C Users Journal"},{"key":"ref11","doi-asserted-by":"crossref","first-page":"488","DOI":"10.21437\/Interspeech.2017-1160","article-title":"Hidden markov model variational autoencoder for acoustic unit discovery","author":"ebbers","year":"2017","journal-title":"InterSpeech"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/N19-4009"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2017-1093"},{"key":"ref32","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3188113"},{"key":"ref2","article-title":"Probabilistic latent semantic analysis","author":"hofmann","year":"1999","journal-title":"Proc Fifteenth Conference on Uncertainty in Artificial Intelligence (UAI&#x2019;99)"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1016\/j.ipm.2018.08.001"},{"key":"ref17","article-title":"Wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"baevski","year":"2020","journal-title":"Adv in NeurIPS"},{"key":"ref39","doi-asserted-by":"publisher","DOI":"10.1002\/(SICI)1097-4571(199009)41:6<391::AID-ASI1>3.0.CO;2-9"},{"key":"ref16","article-title":"Representation learning with contrastive predictive coding","author":"oord","year":"2018"},{"key":"ref38","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10839"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1775"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3207050"},{"key":"ref24","article-title":"The zero resource speech benchmark 2021: Metrics and baselines for unsupervised spoken language modeling","author":"nguyen","year":"2020","journal-title":"Workshop on Self-Supervised Learning for Speech and Audio Processing NeurIPS"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/SLT48900.2021.9383625"},{"key":"ref26","first-page":"1336","article-title":"On generative spoken language modeling from raw audio","volume":"9","author":"lakhotia","year":"2021","journal-title":"Transactions of the Association for Computational Linguistics"},{"key":"ref25","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1755"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU51503.2021.9688137"},{"key":"ref22","first-page":"4171","article-title":"Bert: Pre-training of deep bidirectional transformers for language understanding","author":"kenton","year":"2019","journal-title":"Proc of NAACL-HLT"},{"key":"ref21","article-title":"Unsupervised speech recognition","volume":"34","author":"baevski","year":"2021","journal-title":"Adv in NeurIPS"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1182"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1465"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-1503"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1016\/j.procs.2016.04.033"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2013.05.002"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953257"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/5.880084"},{"key":"ref3","first-page":"993","article-title":"Latent dirichlet allocation","volume":"3","author":"blei","year":"2003","journal-title":"Journal of Machine Learning Research"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2009-559"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-10712"}],"event":{"name":"ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","location":"Rhodes Island, Greece","start":{"date-parts":[[2023,6,4]]},"end":{"date-parts":[[2023,6,10]]}},"container-title":["ICASSP 2023 - 2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10094559\/10094560\/10095280.pdf?arnumber=10095280","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,11,20]],"date-time":"2023-11-20T18:59:05Z","timestamp":1700506745000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10095280\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,4]]},"references-count":39,"URL":"https:\/\/doi.org\/10.1109\/icassp49357.2023.10095280","relation":{},"subject":[],"published":{"date-parts":[[2023,6,4]]}}}