{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,17]],"date-time":"2026-05-17T02:07:20Z","timestamp":1778983640764,"version":"3.51.4"},"publisher-location":"New York, NY, USA","reference-count":31,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,12,14]],"date-time":"2023-12-14T00:00:00Z","timestamp":1702512000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"DOI":"10.13039\/501100006374","name":"National Natural Science Foundation of China","doi-asserted-by":"publisher","award":["62171250, 62301075"],"award-info":[{"award-number":["62171250, 62301075"]}],"id":[{"id":"10.13039\/501100006374","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,12,14]]},"DOI":"10.1145\/3638884.3638917","type":"proceedings-article","created":{"date-parts":[[2024,4,23]],"date-time":"2024-04-23T12:11:26Z","timestamp":1713874286000},"page":"226-230","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":5,"title":["Adversarial Data Augmentation for Robust Speaker Verification"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0009-8492-5222","authenticated-orcid":false,"given":"Zhenyu","family":"Zhou","sequence":"first","affiliation":[{"name":"Beijing University of Post Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0006-9097-0167","authenticated-orcid":false,"given":"Junhui","family":"Chen","sequence":"additional","affiliation":[{"name":"Beijing University of Post Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0004-1504-4345","authenticated-orcid":false,"given":"Namin","family":"Wang","sequence":"additional","affiliation":[{"name":"Huawei Cloud, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5546-8060","authenticated-orcid":false,"given":"Lantian","family":"Li","sequence":"additional","affiliation":[{"name":"Beijing University of Post Telecommunications, China"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-1286-0644","authenticated-orcid":false,"given":"Dong","family":"Wang","sequence":"additional","affiliation":[{"name":"Tsinghua University, China"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2024,4,23]]},"reference":[{"key":"e_1_3_2_1_1_1","doi-asserted-by":"publisher","DOI":"10.1109\/5.628714"},{"key":"e_1_3_2_1_2_1","volume-title":"Speaker recognition by machines and humans: A tutorial review","author":"Hansen HL","year":"2015","unstructured":"[2] John\u00a0HL Hansen and Taufiq Hasan. Speaker recognition by machines and humans: A tutorial review. IEEE Signal processing magazine, 32(6):74\u201399, 2015."},{"key":"e_1_3_2_1_3_1","first-page":"4072","volume-title":"2002 IEEE international conference on acoustics, speech, and signal processing, volume\u00a04","author":"Reynolds A","unstructured":"[3] Douglas\u00a0A Reynolds. An overview of automatic speaker recognition technology. In 2002 IEEE international conference on acoustics, speech, and signal processing, volume\u00a04, pages IV\u20134072. IEEE, 2002."},{"key":"e_1_3_2_1_4_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.08.009"},{"key":"e_1_3_2_1_5_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.neunet.2021.03.004"},{"key":"e_1_3_2_1_6_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"e_1_3_2_1_7_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2205"},{"key":"e_1_3_2_1_8_1","volume-title":"The 2021 NIST speaker recognition evaluation. arXiv preprint arXiv:2204.10242","author":"Sadjadi Seyed\u00a0Omid","year":"2022","unstructured":"[8] Seyed\u00a0Omid Sadjadi, Craig Greenberg, Elliot Singer, Lisa Mason, and Douglas Reynolds. The 2021 NIST speaker recognition evaluation. arXiv preprint arXiv:2204.10242, 2022."},{"key":"e_1_3_2_1_9_1","volume-title":"VoxSRC 2022: The fourth VoxCeleb speaker recognition challenge. arXiv preprint arXiv:2302.10248","author":"Huh Jaesung","year":"2023","unstructured":"[9] Jaesung Huh, Andrew Brown, Jee-weon Jung, Joon\u00a0Son Chung, Arsha Nagrani, Daniel Garcia-Romero, and Andrew Zisserman. VoxSRC 2022: The fourth VoxCeleb speaker recognition challenge. arXiv preprint arXiv:2302.10248, 2023."},{"key":"e_1_3_2_1_10_1","volume-title":"ECAPA-TDNN: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification. arXiv preprint arXiv:2005.07143","author":"Desplanques Brecht","year":"2020","unstructured":"[10] Brecht Desplanques, Jenthe Thienpondt, and Kris Demuynck. ECAPA-TDNN: Emphasized channel attention, propagation and aggregation in tdnn based speaker verification. arXiv preprint arXiv:2005.07143, 2020."},{"key":"e_1_3_2_1_11_1","first-page":"5","volume-title":"2020 28th European Signal Processing Conference (EUSIPCO)","author":"Amini Mohammad\u00a0Mohammad","unstructured":"[11] Mohammad\u00a0Mohammad Amini and Driss Matrouf. Data augmentation versus noise compensation for x-vector speaker recognition systems in noisy environments. In 2020 28th European Signal Processing Conference (EUSIPCO), pages 1\u20135. IEEE, 2021."},{"key":"e_1_3_2_1_12_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7953152"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1508"},{"key":"e_1_3_2_1_14_1","volume-title":"Build a SRE challenge system: Lessons from VoxSRC 2022 and CNSRC","author":"Chen Zhengyang","year":"2022","unstructured":"[14] Zhengyang Chen, Bing Han, Xu\u00a0Xiang, Houjun Huang, Bei Liu, and Yanmin Qian. Build a SRE challenge system: Lessons from VoxSRC 2022 and CNSRC 2022. arXiv preprint arXiv:2211.00815, 2022."},{"key":"e_1_3_2_1_15_1","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003938"},{"key":"e_1_3_2_1_16_1","volume-title":"Specaugment: A simple data augmentation method for automatic speech recognition. arXiv preprint arXiv:1904.08779","author":"Park S","year":"2019","unstructured":"[16] Daniel\u00a0S Park, William Chan, Yu\u00a0Zhang, Chung-Cheng Chiu, Barret Zoph, Ekin\u00a0D Cubuk, and Quoc\u00a0V Le. Specaugment: A simple data augmentation method for automatic speech recognition. arXiv preprint arXiv:1904.08779, 2019."},{"key":"e_1_3_2_1_17_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053481"},{"key":"e_1_3_2_1_18_1","volume-title":"Deep speaker feature learning for text-independent speaker verification. arXiv preprint arXiv:1705.03670","author":"Li Lantian","year":"2017","unstructured":"[18] Lantian Li, Yixiang Chen, Ying Shi, Zhiyuan Tang, and Dong Wang. Deep speaker feature learning for text-independent speaker verification. arXiv preprint arXiv:1705.03670, 2017."},{"key":"e_1_3_2_1_19_1","first-page":"1189","volume-title":"International conference on machine learning","author":"Ganin Yaroslav","unstructured":"[19] Yaroslav Ganin and Victor Lempitsky. Unsupervised domain adaptation by backpropagation. In International conference on machine learning, pages 1180\u20131189. PMLR, 2015."},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461423"},{"key":"e_1_3_2_1_21_1","volume-title":"Augmentation adversarial training for self-supervised speaker recognition. arXiv preprint arXiv:2007.12085","author":"Huh Jaesung","year":"2020","unstructured":"[21] Jaesung Huh, Hee\u00a0Soo Heo, Jingu Kang, Shinji Watanabe, and Joon\u00a0Son Chung. Augmentation adversarial training for self-supervised speaker recognition. arXiv preprint arXiv:2007.12085, 2020."},{"key":"e_1_3_2_1_22_1","doi-asserted-by":"publisher","DOI":"10.1109\/ISCSLP49672.2021.9362053"},{"key":"e_1_3_2_1_23_1","doi-asserted-by":"publisher","DOI":"10.21437\/Odyssey.2022-51"},{"key":"e_1_3_2_1_24_1","volume-title":"Voxceleb2: Deep speaker recognition. arXiv preprint arXiv:1806.05622","author":"Chung Joon\u00a0Son","year":"2018","unstructured":"[24] Joon\u00a0Son Chung, Arsha Nagrani, and Andrew Zisserman. Voxceleb2: Deep speaker recognition. arXiv preprint arXiv:1806.05622, 2018."},{"key":"e_1_3_2_1_25_1","volume-title":"MUSAN: A music, speech, and noise corpus. arXiv preprint arXiv:1510.08484","author":"Snyder David","year":"2015","unstructured":"[25] David Snyder, Guoguo Chen, and Daniel Povey. MUSAN: A music, speech, and noise corpus. arXiv preprint arXiv:1510.08484, 2015."},{"key":"e_1_3_2_1_26_1","volume-title":"THCHS-30: A free Chinese speech corpus. arXiv preprint arXiv:1512.01882","author":"Wang Dong","year":"2015","unstructured":"[26] Dong Wang and Xuewei Zhang. THCHS-30: A free Chinese speech corpus. arXiv preprint arXiv:1512.01882, 2015."},{"key":"e_1_3_2_1_27_1","first-page":"7608","volume-title":"ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP)","author":"Fan Yue","unstructured":"[27] Yue Fan, JW\u00a0Kang, LT\u00a0Li, KC\u00a0Li, HL\u00a0Chen, ST\u00a0Cheng, PY\u00a0Zhang, ZY\u00a0Zhou, YQ\u00a0Cai, and Dong Wang. CN-Celeb: a challenging Chinese speaker recognition dataset. In ICASSP 2020-2020 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pages 7604\u20137608. IEEE, 2020."},{"key":"e_1_3_2_1_28_1","first-page":"1656","volume-title":"APSIPA ASC","author":"Wang Shuai","unstructured":"[28] Xu\u00a0Xiang, Shuai Wang, Houjun Huang, Yanmin Qian, and Kai Yu. Margin matters: Towards more discriminative deep neural network embeddings for speaker recognition. In APSIPA ASC, pages 1652\u20131656. IEEE, 2019."},{"key":"e_1_3_2_1_29_1","volume-title":"Attentive statistics pooling for deep speaker embedding. arXiv preprint arXiv:1803.10963","author":"Okabe Koji","year":"2018","unstructured":"[29] Koji Okabe, Takafumi Koshinaka, and Koichi Shinoda. Attentive statistics pooling for deep speaker embedding. arXiv preprint arXiv:1803.10963, 2018."},{"key":"e_1_3_2_1_30_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2017.7952261"},{"key":"e_1_3_2_1_31_1","volume-title":"Spot keywords from very noisy and mixed speech. arXiv preprint arXiv:2305.17706","author":"Shi Ying","year":"2023","unstructured":"[31] Ying Shi, Dong Wang, Lantian Li, Jiqing Han, and Shi Yin. Spot keywords from very noisy and mixed speech. arXiv preprint arXiv:2305.17706, 2023."}],"event":{"name":"ICCIP 2023: 2023 the 9th International Conference on Communication and Information Processing","location":"Lingshui China","acronym":"ICCIP 2023"},"container-title":["Proceedings of the 2023 9th International Conference on Communication and Information Processing"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3638884.3638917","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3638884.3638917","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,5,17]],"date-time":"2026-05-17T01:47:34Z","timestamp":1778982454000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3638884.3638917"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,12,14]]},"references-count":31,"alternative-id":["10.1145\/3638884.3638917","10.1145\/3638884"],"URL":"https:\/\/doi.org\/10.1145\/3638884.3638917","relation":{},"subject":[],"published":{"date-parts":[[2023,12,14]]},"assertion":[{"value":"2024-04-23","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}