{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,5,8]],"date-time":"2025-05-08T04:48:18Z","timestamp":1746679698870,"version":"3.28.0"},"reference-count":30,"publisher":"IEEE","license":[{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2023,6,4]],"date-time":"2023-06-04T00:00:00Z","timestamp":1685836800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2023,6,4]]},"DOI":"10.1109\/icasspw59220.2023.10192957","type":"proceedings-article","created":{"date-parts":[[2023,8,2]],"date-time":"2023-08-02T17:30:54Z","timestamp":1690997454000},"page":"1-5","source":"Crossref","is-referenced-by-count":4,"title":["Improving Dino-Based Self-Supervised Speaker Verification with Progressive Cluster-Aware Training"],"prefix":"10.1109","author":[{"given":"Bing","family":"Han","sequence":"first","affiliation":[{"name":"Shanghai Jiao Tong University,MoE Key Lab of Artificial Intelligence, AI Institute X-LANCE Lab,Department of Computer Science and Engineering,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Wen","family":"Huang","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University,MoE Key Lab of Artificial Intelligence, AI Institute X-LANCE Lab,Department of Computer Science and Engineering,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Zhengyang","family":"Chen","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University,MoE Key Lab of Artificial Intelligence, AI Institute X-LANCE Lab,Department of Computer Science and Engineering,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yanmin","family":"Qian","sequence":"additional","affiliation":[{"name":"Shanghai Jiao Tong University,MoE Key Lab of Artificial Intelligence, AI Institute X-LANCE Lab,Department of Computer Science and Engineering,Shanghai,China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref13","article-title":"Augmentation adversarial training for unsupervised speaker recognition","author":"huh","year":"2020","journal-title":"arXiv preprint arXiv 2007 12869"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9747814"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9414973"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413351"},{"key":"ref30","first-page":"1641","article-title":"Semi-supervised contrastive learning with generalized contrastive loss and its application to speaker recognition","author":"inoue","year":"2020","journal-title":"Proc IEEE APSIPA ASC"},{"key":"ref11","article-title":"Exploring wav2vec 2.0 on speaker verification and language identification","author":"fan","year":"2020","journal-title":"2012 arXiv preprint arXiv"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"ref2","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICCV48922.2021.00951"},{"key":"ref16","article-title":"Unsupervised representation learning for speaker recognition via contrastive equilibrium learning","author":"mun","year":"2020","journal-title":"arXiv preprint arXiv 2010 14318"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3197315"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-742"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/ICPR48806.2021.9412301"},{"key":"ref26","first-page":"2616","article-title":"Voxceleb: A large-scale speaker identification dataset","author":"nagrani","year":"2017","journal-title":"Proc ISCA Interspeech"},{"key":"ref25","article-title":"Musan: A music, speech, and noise corpus","author":"snyder","year":"2015","journal-title":"arXiv preprint arXiv 1510 08484"},{"key":"ref20","article-title":"Self-supervised curriculum learning for speaker verification","author":"heo","year":"2022","journal-title":"arXiv preprint arXiv 2203 14525"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/SLT54892.2023.10022470"},{"key":"ref21","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2022.3198315"},{"key":"ref28","first-page":"6829","article-title":"Disentangled speech embeddings using crossmodal self-supervision","author":"nagrani","year":"2020","journal-title":"Proc IEEE ICASSP"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054017"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-1113"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2842"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746178"},{"key":"ref9","first-page":"12449","article-title":"wav2vec 2.0: A framework for self-supervised learning of speech representations","volume":"33","author":"baevski","year":"2020","journal-title":"Proc NIPS"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/APSIPAASC47483.2019.9023039"},{"key":"ref3","article-title":"But system description to voxceleb speaker recognition challenge 2019","author":"zeinali","year":"2019","journal-title":"arXiv preprint arXiv 1910 12731"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP49672.2021.9362097"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2016.7472652"}],"event":{"name":"2023 IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops (ICASSPW)","start":{"date-parts":[[2023,6,4]]},"location":"Rhodes Island, Greece","end":{"date-parts":[[2023,6,10]]}},"container-title":["2023 IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops (ICASSPW)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/10192576\/10192577\/10192957.pdf?arnumber=10192957","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,8,21]],"date-time":"2023-08-21T17:42:20Z","timestamp":1692639740000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/10192957\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,6,4]]},"references-count":30,"URL":"https:\/\/doi.org\/10.1109\/icasspw59220.2023.10192957","relation":{},"subject":[],"published":{"date-parts":[[2023,6,4]]}}}