{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,6,19]],"date-time":"2026-06-19T15:48:07Z","timestamp":1781884087051,"version":"3.54.5"},"publisher-location":"New York, NY, USA","reference-count":60,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,4,19]],"date-time":"2023-04-19T00:00:00Z","timestamp":1681862400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"funder":[{"name":"JST Moonshot","award":["JPMJMS2012"],"award-info":[{"award-number":["JPMJMS2012"]}]},{"name":"JST CREST","award":["JPMJCR17A3"],"award-info":[{"award-number":["JPMJCR17A3"]}]}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,4,19]]},"DOI":"10.1145\/3544548.3580706","type":"proceedings-article","created":{"date-parts":[[2023,4,20]],"date-time":"2023-04-20T04:27:55Z","timestamp":1681964875000},"page":"1-12","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":27,"title":["WESPER: Zero-shot and Realtime Whisper to Normal Voice Conversion for Whisper-based Speech Interactions"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3629-2514","authenticated-orcid":false,"given":"Jun","family":"Rekimoto","sequence":"first","affiliation":[{"name":"The University of Tokyo, Japan and Sony CSL Kyoto, Japan"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"320","published-online":{"date-parts":[[2023,4,19]]},"reference":[{"key":"e_1_3_3_2_1_1","doi-asserted-by":"publisher","unstructured":"A. Al-Nasheri G. Muhammad M. Alsulaiman and Z. Ali. 2017. Investigation of Voice Pathology Detection and Classification on Different Frequency Regions Using Correlation Functions. Journal of Voice (2017). https:\/\/doi.org\/10.1016\/j.jvoice.2016.01.014","DOI":"10.1016\/j.jvoice.2016.01.014"},{"key":"e_1_3_3_2_2_1","volume-title":"wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. arXiv [cs.CL] (June","author":"Baevski Alexei","year":"2020","unstructured":"Alexei Baevski, Henry Zhou, Abdelrahman Mohamed, and Michael Auli. 2020. wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. arXiv [cs.CL] (June 2020)."},{"key":"e_1_3_3_2_3_1","doi-asserted-by":"publisher","DOI":"10.1109\/MC.2015.310"},{"key":"e_1_3_3_2_4_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1904.04169"},{"key":"e_1_3_3_2_5_1","doi-asserted-by":"publisher","unstructured":"Zal\u00e1n Borsos Rapha\u00ebl Marinier Damien Vincent Eugene Kharitonov Olivier Pietquin Matt Sharifi Olivier Teboul David Grangier Marco Tagliasacchi and Neil Zeghidour. 2022. AudioLM: a Language Modeling Approach to Audio Generation. https:\/\/doi.org\/10.48550\/ARXIV.2209.03143","DOI":"10.48550\/ARXIV.2209.03143"},{"key":"e_1_3_3_2_6_1","volume-title":"End-to-end Whispered Speech Recognition with Frequency-weighted Approaches and Pseudo Whisper Pre-training. (May","author":"Chang Heng-Jui","year":"2020","unstructured":"Heng-Jui Chang, Alexander\u00a0H Liu, Hung-Yi Lee, and Lin-Shan Lee. 2020. End-to-end Whispered Speech Recognition with Frequency-weighted Approaches and Pseudo Whisper Pre-training. (May 2020). arxiv:2005.01972\u00a0[cs.CL]"},{"key":"e_1_3_3_2_7_1","doi-asserted-by":"publisher","DOI":"10.1145\/3498361.3538933"},{"key":"e_1_3_3_2_8_1","doi-asserted-by":"publisher","unstructured":"Chung-Ming Chien Jheng-Hao Lin Chien-yu Huang Po-chun Hsu and Hung-yi Lee. 2021. Investigating on Incorporating Pretrained and Learnable Speaker Representations for Multi-Speaker Multi-Style Text-to-Speech. In ICASSP 2021 - 2021 IEEE International Conference on Acoustics Speech and Signal Processing (ICASSP). 8588\u20138592. https:\/\/doi.org\/10.1109\/ICASSP39728.2021.9413880","DOI":"10.1109\/ICASSP39728.2021.9413880"},{"key":"e_1_3_3_2_9_1","doi-asserted-by":"publisher","unstructured":"Alexis Conneau Alexei Baevski Ronan Collobert Abdelrahman Mohamed and Michael Auli. 2020. Unsupervised Cross-lingual Representation Learning for Speech Recognition. https:\/\/doi.org\/10.48550\/ARXIV.2006.13979","DOI":"10.48550\/ARXIV.2006.13979"},{"key":"e_1_3_3_2_10_1","volume-title":"Voice Conversion for Whispered Speech Synthesis. (Dec","author":"Cotescu Marius","year":"2019","unstructured":"Marius Cotescu, Thomas Drugman, Goeric Huybrechts, Jaime Lorenzo-Trueba, and Alexis Moinet. 2019. Voice Conversion for Whispered Speech Synthesis. (Dec. 2019). arxiv:1912.05289\u00a0[cs.SD]"},{"key":"e_1_3_3_2_11_1","doi-asserted-by":"publisher","DOI":"10.1016\/j.specom.2009.08.002"},{"key":"e_1_3_3_2_12_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1810.04805"},{"key":"e_1_3_3_2_13_1","unstructured":"Dyson. 2022. dyson zone: Air-purifying headphones with active noise cancelling. https:\/\/www.dyson.co.uk\/en."},{"key":"e_1_3_3_2_14_1","volume-title":"An Introduction to Silent Speech Interfaces","author":"Freitas Joo","unstructured":"Joo Freitas, Antnio Teixeira, Miguel\u00a0Sales Dias, and Samuel Silva. 2016. An Introduction to Silent Speech Interfaces (1st ed.). Springer Publishing Company, Incorporated.","edition":"1"},{"key":"e_1_3_3_2_15_1","doi-asserted-by":"publisher","DOI":"10.1145\/3242587.3242603"},{"key":"e_1_3_3_2_16_1","doi-asserted-by":"publisher","unstructured":"Teng Gao Jian Zhou Huabin Wang Liang Tao and Hon\u00a0Keung Kwan. 2021. Attention-Guided Generative Adversarial Network for Whisper to Normal Speech Conversion. https:\/\/doi.org\/10.48550\/ARXIV.2111.01342","DOI":"10.48550\/ARXIV.2111.01342"},{"key":"e_1_3_3_2_17_1","doi-asserted-by":"publisher","DOI":"10.1145\/1143844.1143891"},{"key":"e_1_3_3_2_18_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2017.2738559"},{"key":"e_1_3_3_2_19_1","doi-asserted-by":"publisher","unstructured":"Tomoki Hayashi Wen-Chin Huang Kazuhiro Kobayashi and Tomoki Toda. 2021. Non-autoregressive sequence-to-sequence voice conversion. https:\/\/doi.org\/10.48550\/ARXIV.2104.06793","DOI":"10.48550\/ARXIV.2104.06793"},{"key":"e_1_3_3_2_20_1","doi-asserted-by":"publisher","DOI":"10.1145\/1107548.1107577"},{"key":"e_1_3_3_2_21_1","doi-asserted-by":"publisher","DOI":"10.1109\/TASLP.2021.3122291"},{"key":"e_1_3_3_2_22_1","doi-asserted-by":"publisher","unstructured":"Wen-Chin Huang Tomoki Hayashi Shinji Watanabe and Tomoki Toda. 2020. The Sequence-to-Sequence Baseline for the Voice Conversion Challenge 2020: Cascading ASR and TTS. https:\/\/doi.org\/10.48550\/ARXIV.2010.02434","DOI":"10.48550\/ARXIV.2010.02434"},{"key":"e_1_3_3_2_23_1","unstructured":"Amazon.com Inc.2018. How Alexa keeps getting smarter. https:\/\/www.aboutamazon.com\/devices\/how-alexa-keeps-getting-smarter"},{"key":"e_1_3_3_2_24_1","unstructured":"Google Inc.2020. Google Cloud Speech-to-Text. https:\/\/cloud.google.com\/speech-to-text."},{"key":"e_1_3_3_2_25_1","unstructured":"Prolific inc.2014. Prolific. https:\/\/www.prolific.co"},{"key":"e_1_3_3_2_26_1","unstructured":"Philips Inc.2021. Fresh Air Mask Series 6000. https:\/\/www.philips.com.sg\/c-p\/ACM066_01\/fresh-air-mask-series-6000."},{"key":"e_1_3_3_2_27_1","unstructured":"Keith Ito and Linda Johnson. 2017. The LJ Speech Dataset. https:\/\/keithito.com\/LJ-Speech-Dataset\/."},{"key":"e_1_3_3_2_28_1","unstructured":"jfsantos. 2019. mushraJS. https:\/\/github.com\/jfsantos\/mushraJS"},{"key":"e_1_3_3_2_29_1","doi-asserted-by":"publisher","unstructured":"Hirokazu Kameoka Takuhiro Kaneko Kou Tanaka and Nobukatsu Hojo. 2018. StarGAN-VC: Non-parallel many-to-many voice conversion with star generative adversarial networks. https:\/\/doi.org\/10.48550\/ARXIV.1806.02169","DOI":"10.48550\/ARXIV.1806.02169"},{"key":"e_1_3_3_2_30_1","doi-asserted-by":"publisher","unstructured":"Takuhiro Kaneko and Hirokazu Kameoka. 2017. Parallel-Data-Free Voice Conversion Using Cycle-Consistent Adversarial Networks. https:\/\/doi.org\/10.48550\/ARXIV.1711.11293","DOI":"10.48550\/ARXIV.1711.11293"},{"key":"e_1_3_3_2_31_1","doi-asserted-by":"publisher","DOI":"10.1145\/3172944.3172977"},{"key":"e_1_3_3_2_32_1","doi-asserted-by":"publisher","DOI":"10.1145\/3290605.3300376"},{"key":"e_1_3_3_2_33_1","doi-asserted-by":"publisher","unstructured":"Jungil Kong Jaehyeon Kim and Jaekyoung Bae. 2020. HiFi-GAN: Generative Adversarial Networks for Efficient and High Fidelity Speech Synthesis. https:\/\/doi.org\/10.48550\/ARXIV.2010.05646","DOI":"10.48550\/ARXIV.2010.05646"},{"key":"e_1_3_3_2_34_1","doi-asserted-by":"publisher","unstructured":"Felix Kreuk Adam Polyak Jade Copet Eugene Kharitonov Tu-Anh Nguyen Morgane Rivi\u00e8re Wei-Ning Hsu Abdelrahman Mohamed Emmanuel Dupoux and Yossi Adi. 2021. Textless Speech Emotion Conversion using Discrete and Decomposed Representations. https:\/\/doi.org\/10.48550\/ARXIV.2111.07402","DOI":"10.48550\/ARXIV.2111.07402"},{"key":"e_1_3_3_2_35_1","doi-asserted-by":"publisher","unstructured":"Kushal Lakhotia Evgeny Kharitonov Wei-Ning Hsu Yossi Adi Adam Polyak Benjamin Bolte Tu-Anh Nguyen Jade Copet Alexei Baevski Adelrahman Mohamed and Emmanuel Dupoux. 2021. Generative Spoken Language Modeling from Raw Audio. https:\/\/doi.org\/10.48550\/ARXIV.2102.01192","DOI":"10.48550\/ARXIV.2102.01192"},{"key":"e_1_3_3_2_37_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9052966"},{"key":"e_1_3_3_2_38_1","doi-asserted-by":"publisher","unstructured":"Michael McAuliffe Michaela Socolof Sarah Mihuc Michael Wagner and Morgan Sonderegger. 2017. Montreal Forced Aligner: Trainable Text-Speech Alignment Using Kaldi. 498\u2013502. https:\/\/doi.org\/10.21437\/Interspeech.2017-1386","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"e_1_3_3_2_39_1","doi-asserted-by":"publisher","DOI":"10.48550\/ARXIV.1802.03426"},{"key":"e_1_3_3_2_40_1","doi-asserted-by":"publisher","unstructured":"Lisa\u00a0Lucks Mendel Sungmin Lee Monique Pousson Chhayakanta Patro Skylar McSorley Bonny Banerjee Shamima Najnin and Masoumeh\u00a0Heidari Kapourchali. 2017. Corpus of deaf speech for acoustic and speech production research. The Journal of the Acoustical Society of America 142 (1)(2017) EL102. https:\/\/doi.org\/10.1121\/1.4994288","DOI":"10.1121\/1.4994288"},{"key":"e_1_3_3_2_41_1","doi-asserted-by":"publisher","DOI":"10.24840\/2183-6493_008.002_0016"},{"key":"e_1_3_3_2_42_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"e_1_3_3_2_43_1","doi-asserted-by":"publisher","unstructured":"Santiago Pascual Antonio Bonafonte Joan Serr\u00e0 and Jose\u00a0A. Gonzalez. 2018. Whispered-to-voiced Alaryngeal Speech Conversion with Generative Adversarial Networks. https:\/\/doi.org\/10.48550\/ARXIV.1808.10687","DOI":"10.48550\/ARXIV.1808.10687"},{"key":"e_1_3_3_2_44_1","volume-title":"Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings. (April","author":"Pepino Leonardo","year":"2021","unstructured":"Leonardo Pepino, Pablo Riera, and Luciana Ferrer. 2021. Emotion Recognition from Speech Using Wav2vec 2.0 Embeddings. (April 2021). arxiv:2104.03502\u00a0[cs.SD]"},{"key":"e_1_3_3_2_45_1","unstructured":"Manfred P\u030eutzer and William\u00a0J. Barry. 2016. Saarbruecken voice database. https:\/\/stimmdb.coli.uni-saarland.de"},{"key":"e_1_3_3_2_46_1","volume-title":"Proceedings of the 36th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a097)","author":"Qian Kaizhi","year":"2019","unstructured":"Kaizhi Qian, Yang Zhang, Shiyu Chang, Xuesong Yang, and Mark Hasegawa-Johnson. 2019. AutoVC: Zero-Shot Voice Style Transfer with Only Autoencoder Loss. In Proceedings of the 36th International Conference on Machine Learning(Proceedings of Machine Learning Research, Vol.\u00a097), Kamalika Chaudhuri and Ruslan Salakhutdinov (Eds.). PMLR, 5210\u20135219. https:\/\/proceedings.mlr.press\/v97\/qian19c.html"},{"key":"e_1_3_3_2_47_1","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.1911.02637"},{"key":"e_1_3_3_2_48_1","doi-asserted-by":"publisher","DOI":"10.1145\/3526113.3545685"},{"key":"e_1_3_3_2_49_1","doi-asserted-by":"publisher","unstructured":"Yi Ren Chenxu Hu Xu Tan Tao Qin Sheng Zhao Zhou Zhao and Tie-Yan Liu. 2020. FastSpeech 2: Fast and High-Quality End-to-End Text to Speech. https:\/\/doi.org\/10.48550\/ARXIV.2006.04558","DOI":"10.48550\/ARXIV.2006.04558"},{"key":"e_1_3_3_2_50_1","doi-asserted-by":"publisher","DOI":"10.1145\/2971763.2971765"},{"key":"e_1_3_3_2_51_1","doi-asserted-by":"publisher","DOI":"10.1145\/2634317.2634322"},{"key":"e_1_3_3_2_52_1","unstructured":"seeed studio. 2019. ReSpeaker USB Mic Array. https:\/\/wiki.seeedstudio.com\/ReSpeaker-USB-Mic-Array\/"},{"key":"e_1_3_3_2_53_1","volume-title":"Novel MMSE DiscoGAN for Cross-Domain Whisper-to-Speech Conversion. Machine Learning. In Speech and Language Processing (MLSLP) Workshop.","author":"Shah Nirmesh","year":"2018","unstructured":"Nirmesh Shah, Mihir Parmar, Neil Shah, and Hemant Patil. 2018. Novel MMSE DiscoGAN for Cross-Domain Whisper-to-Speech Conversion. Machine Learning. In Speech and Language Processing (MLSLP) Workshop."},{"key":"e_1_3_3_2_54_1","doi-asserted-by":"publisher","DOI":"10.1145\/3242587.3242599"},{"key":"e_1_3_3_2_55_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICME.2016.7552917"},{"key":"e_1_3_3_2_56_1","unstructured":"International\u00a0Telecommunication Union. 2013. BS.1534 : Method for the subjective assessment of intermediate quality level of audio systems. https:\/\/www.itu.int\/rec\/R-REC-BS.1534\/en"},{"key":"e_1_3_3_2_57_1","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP43922.2022.9746484"},{"key":"e_1_3_3_2_58_1","volume-title":"Tacotron: A Fully End-to-End Text-To-Speech Synthesis Model. CoRR abs\/1703.10135(2017). arXiv:1703.10135http:\/\/arxiv.org\/abs\/1703.10135","author":"Wang Yuxuan","year":"2017","unstructured":"Yuxuan Wang, R.\u00a0J. Skerry-Ryan, Daisy Stanton, Yonghui Wu, Ron\u00a0J. Weiss, Navdeep Jaitly, Zongheng Yang, Ying Xiao, Zhifeng Chen, Samy Bengio, Quoc\u00a0V. Le, Yannis Agiomyrgiannakis, Rob Clark, and Rif\u00a0A. Saurous. 2017. Tacotron: A Fully End-to-End Text-To-Speech Synthesis Model. CoRR abs\/1703.10135(2017). arXiv:1703.10135http:\/\/arxiv.org\/abs\/1703.10135"},{"key":"e_1_3_3_2_59_1","unstructured":"Cheng Yi Jianzhong Wang Ning Cheng Shiyu Zhou and Bo Xu. 2020. Applying Wav2vec2.0 to Speech Recognition in Various Low-resource Languages. (Dec. 2020). arxiv:2012.12121\u00a0[cs.CL]"},{"key":"e_1_3_3_2_60_1","unstructured":"zeta chicken. 2017. toWhisper. https:\/\/github.com\/zeta-chicken\/toWhisper"},{"key":"e_1_3_3_2_61_1","doi-asserted-by":"publisher","unstructured":"Qiu-Shi Zhu Long Zhou Jie Zhang Shu-Jie Liu Yu-Chen Hu and Li-Rong Dai. 2022. Robust Data2vec: Noise-robust Speech Representation Learning for ASR by Combining Regression and Improved Contrastive Learning. https:\/\/doi.org\/10.48550\/ARXIV.2210.15324","DOI":"10.48550\/ARXIV.2210.15324"}],"event":{"name":"CHI '23: CHI Conference on Human Factors in Computing Systems","location":"Hamburg Germany","acronym":"CHI '23","sponsor":["SIGCHI ACM Special Interest Group on Computer-Human Interaction"]},"container-title":["Proceedings of the 2023 CHI Conference on Human Factors in Computing Systems"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3544548.3580706","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3544548.3580706","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T16:37:24Z","timestamp":1750178244000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3544548.3580706"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,4,19]]},"references-count":60,"alternative-id":["10.1145\/3544548.3580706","10.1145\/3544548"],"URL":"https:\/\/doi.org\/10.1145\/3544548.3580706","relation":{},"subject":[],"published":{"date-parts":[[2023,4,19]]},"assertion":[{"value":"2023-04-19","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}