{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,11,29]],"date-time":"2025-11-29T07:06:42Z","timestamp":1764400002060,"version":"3.46.0"},"reference-count":36,"publisher":"IEEE","license":[{"start":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T00:00:00Z","timestamp":1761091200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2025,10,22]],"date-time":"2025-10-22T00:00:00Z","timestamp":1761091200000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2025,10,22]]},"DOI":"10.1109\/apsipaasc65261.2025.11248997","type":"proceedings-article","created":{"date-parts":[[2025,11,28]],"date-time":"2025-11-28T18:40:26Z","timestamp":1764355226000},"page":"2063-2068","source":"Crossref","is-referenced-by-count":0,"title":["Voice Privacy Protection with Adversarial Examples Using Anchor Speaker Embedding"],"prefix":"10.1109","author":[{"given":"Shunya","family":"Ishikawa","sequence":"first","affiliation":[{"name":"The University of Electro-Communications,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Yuki","family":"Katsumata","sequence":"additional","affiliation":[{"name":"The University of Electro-Communications,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Toru","family":"Nakashika","sequence":"additional","affiliation":[{"name":"The University of Electro-Communications,Japan"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"263","reference":[{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.18653\/v1\/2024.acl-long.673"},{"key":"ref2","first-page":"22605","article-title":"NaturalSpeech 3: Zero-Shot Speech Synthesis with Factorized Codec and Diffusion Models","volume-title":"Proc. ICML","author":"Ju","year":"2024"},{"key":"ref3","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.04.004"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1016\/j.jnca.2024.103989"},{"key":"ref5","article-title":"Sok: Comprehensive Security Overview, Challenges, and Future Directions of Voice-Controlled Systems","author":"Xu","year":"2024","journal-title":"arXiv"},{"key":"ref6","article-title":"OpenVoice: Versatile Instant Voice Cloning","author":"Qin","year":"2024","journal-title":"arXiv"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/TASLPRO.2025.3530270"},{"key":"ref8","first-page":"2709","article-title":"YourTTS: Towards Zero-Shot Multi-Speaker TTS and Zero-Shot Voice Conversion for Everyone","volume":"162","author":"Casanova","year":"2022","journal-title":"Proc. ICML"},{"key":"ref9","article-title":"Guided-TTS2: A Diffusion Model for High-Quality Adaptive Text-to-Speech with Untranscribed Data","author":"Kim","year":"2022","journal-title":"arXiv"},{"key":"ref10","article-title":"Explaining and Harnessing Adversarial Examples","author":"Goodfellow","year":"2015","journal-title":"Proc. ICLR"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.1201\/9781351251389-8"},{"key":"ref12","article-title":"Adversarial Machine Learning at Scale","author":"Kurakin","year":"2017","journal-title":"Proc. ICLR"},{"key":"ref13","article-title":"Intriguing Properties of Neural Networks","author":"Szegedy","year":"2014","journal-title":"Proc. ICLR"},{"key":"ref14","doi-asserted-by":"publisher","DOI":"10.1109\/CVPR.2019.00284"},{"key":"ref15","doi-asserted-by":"publisher","DOI":"10.1109\/cvpr.2019.00444"},{"key":"ref16","article-title":"Delving into Transferable Adversarial Examples and Black-Box Attacks","author":"Liu","year":"2017","journal-title":"Proc. ICLR"},{"key":"ref17","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746806"},{"key":"ref18","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"ref21","article-title":"Sequence Transduction with Recurrent Neural Networks","author":"Graves","year":"2012","journal-title":"Proc. ICML Workshop on Representation Learning"},{"key":"ref22","doi-asserted-by":"publisher","DOI":"10.1109\/JSTSP.2017.2763455"},{"key":"ref23","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639038"},{"key":"ref24","article-title":"Crafting Adversarial Examples for Speech Paralinguistics Applications","author":"Gong","year":"2018","journal-title":"Proc. DYNAMICS Workshop"},{"key":"ref25","first-page":"6980","article-title":"Houdini: Fooling Deep Structured Prediction Models","author":"Cisse","year":"2017","journal-title":"Proc. NIPS"},{"key":"ref26","first-page":"49","article-title":"CommanderSong: A Systematic Approach for Practical Adversarial Voice Recognition","author":"Yuan","year":"2018","journal-title":"Proc. USENIX Security"},{"key":"ref27","first-page":"5231","article-title":"Imperceptible, Robust, and Targeted Adversar-ial Examples for Automatic Speech Recognition","author":"Qin","year":"2019","journal-title":"Proc. ICML"},{"key":"ref28","doi-asserted-by":"publisher","DOI":"10.1109\/SPW.2018.00009"},{"key":"ref29","doi-asserted-by":"publisher","DOI":"10.1109\/DSC55868.2022.00071"},{"key":"ref30","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP48485.2024.10447699"},{"key":"ref31","first-page":"5530","article-title":"Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech","author":"Kim","year":"2021","journal-title":"Proc. ICML"},{"key":"ref32","article-title":"Adam: A Method for Stochastic Optimization","author":"Kingma","year":"2015","journal-title":"Proc. ICLR"},{"key":"ref33","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2441"},{"key":"ref34","article-title":"Decoupled Weight Decay Regularization","author":"Loshchilov","year":"2019","journal-title":"Proc. ICLR"},{"key":"ref35","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2022-439"},{"key":"ref36","article-title":"SpeechBrain: A General-Purpose Speech Toolkit","author":"Ravanelli","year":"2021","journal-title":"arXiv"}],"event":{"name":"2025 Asia Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC)","start":{"date-parts":[[2025,10,22]]},"location":"Singapore, Singapore","end":{"date-parts":[[2025,10,24]]}},"container-title":["2025 Asia Pacific Signal and Information Processing Association Annual Summit and Conference (APSIPA ASC)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx8\/11248853\/11248968\/11248997.pdf?arnumber=11248997","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,11,29]],"date-time":"2025-11-29T07:02:31Z","timestamp":1764399751000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/11248997\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,10,22]]},"references-count":36,"URL":"https:\/\/doi.org\/10.1109\/apsipaasc65261.2025.11248997","relation":{},"subject":[],"published":{"date-parts":[[2025,10,22]]}}}