{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2025,6,18]],"date-time":"2025-06-18T04:12:29Z","timestamp":1750219949836,"version":"3.41.0"},"publisher-location":"New York, NY, USA","reference-count":22,"publisher":"ACM","license":[{"start":{"date-parts":[[2023,7,5]],"date-time":"2023-07-05T00:00:00Z","timestamp":1688515200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.acm.org\/publications\/policies\/copyright_policy#Background"}],"content-domain":{"domain":["dl.acm.org"],"crossmark-restriction":true},"short-container-title":[],"published-print":{"date-parts":[[2023,7,5]]},"DOI":"10.1145\/3594806.3596536","type":"proceedings-article","created":{"date-parts":[[2023,8,10]],"date-time":"2023-08-10T17:38:12Z","timestamp":1691689092000},"page":"700-706","update-policy":"https:\/\/doi.org\/10.1145\/crossmark-policy","source":"Crossref","is-referenced-by-count":4,"title":["Practical Study of Deep Learning Models for Speech Synthesis"],"prefix":"10.1145","author":[{"ORCID":"https:\/\/orcid.org\/0009-0006-7135-3809","authenticated-orcid":false,"given":"Quentin","family":"Langlois","sequence":"first","affiliation":[{"name":"UCLouvain, Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0001-6685-7398","authenticated-orcid":false,"given":"S\u00e9bastien","family":"Jodogne","sequence":"additional","affiliation":[{"name":"UCLouvain, Belgium"}],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"320","published-online":{"date-parts":[[2023,8,10]]},"reference":[{"key":"e_1_3_2_1_1_1","volume-title":"Common Voice: A Massively-Multilingual Speech Corpus. CoRR abs\/1912.06670","author":"Ardila Rosana","year":"2019","unstructured":"Rosana Ardila, Megan Branson, Kelly Davis, Michael Henretty, Michael Kohler, Josh Meyer, Reuben Morais, Lindsay Saunders, Francis\u00a0M. Tyers, and Gregor Weber. 2019. Common Voice: A Massively-Multilingual Speech Corpus. CoRR abs\/1912.06670 (2019), 5\u00a0pages. http:\/\/arxiv.org\/abs\/1912.06670"},{"volume-title":"Smart Computing Paradigms: New Progresses and Challenges, Atilla El\u00e7i, Pankaj\u00a0Kumar Sa, Chirag\u00a0N","author":"Baruah Ujwala","key":"e_1_3_2_1_2_1","unstructured":"Ujwala Baruah, Rabul\u00a0Hussain Laskar, and Biswajit Purkayashtha. 2020. Speaker Verification Systems: A Comprehensive Review. In Smart Computing Paradigms: New Progresses and Challenges, Atilla El\u00e7i, Pankaj\u00a0Kumar Sa, Chirag\u00a0N. Modi, Gustavo Olague, Manmath\u00a0N. Sahoo, and Sambit Bakshi (Eds.). Springer Singapore, Singapore, 195\u2013207."},{"key":"e_1_3_2_1_3_1","doi-asserted-by":"publisher","DOI":"10.1162\/neco.1997.9.8.1735"},{"key":"e_1_3_2_1_4_1","unstructured":"Keith Ito. 2017. The LJ Speech Dataset. https:\/\/keithito.com\/LJ-Speech-Dataset\/."},{"key":"e_1_3_2_1_5_1","unstructured":"Corentin Jemine. 2019. Real-Time Voice Cloning. Master\u2019s thesis. ULi\u00e8ge. https:\/\/matheo.uliege.be\/handle\/2268.2\/6801"},{"key":"e_1_3_2_1_6_1","volume-title":"Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis. CoRR abs\/1806.04558","author":"Jia Ye","year":"2018","unstructured":"Ye Jia, Yu Zhang, Ron\u00a0J. Weiss, Quan Wang, Jonathan Shen, Fei Ren, Zhifeng Chen, Patrick Nguyen, Ruoming Pang, Ignacio Lopez-Moreno, and Yonghui Wu. 2018. Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis. CoRR abs\/1806.04558 (2018), 15\u00a0pages. arXiv:1806.04558http:\/\/arxiv.org\/abs\/1806.04558"},{"key":"e_1_3_2_1_7_1","volume-title":"Kingma and Prafulla Dhariwal","author":"P.","year":"2018","unstructured":"Diederik\u00a0P. Kingma and Prafulla Dhariwal. 2018. Glow: Generative Flow with Invertible 1x1 Convolutions. ArXiv abs\/1807.03039 (2018), 15\u00a0pages. https:\/\/arxiv.org\/abs\/1807.03039"},{"key":"e_1_3_2_1_8_1","unstructured":"Gregory Koch Richard Zemel Ruslan Salakhutdinov 2015. Siamese Neural Networks for One-Shot Image Recognition. In ICML deep learning workshop Vol.\u00a02. W&CP Lille 8\u00a0pages."},{"key":"e_1_3_2_1_9_1","unstructured":"Ken MacLean. 2018. Voxforge. http:\/\/www.voxforge.org\/home"},{"key":"e_1_3_2_1_10_1","unstructured":"NVIDIA. 2020. Tacotron 2 (without wavenet). https:\/\/github.com\/NVIDIA\/tacotron2"},{"key":"e_1_3_2_1_11_1","unstructured":"Kyubyong Park and Tommy Mulc. 2018. French Single Speaker Speech Dataset. https:\/\/www.kaggle.com\/datasets\/bryanpark\/french-single-speaker-speech-dataset"},{"key":"e_1_3_2_1_12_1","volume-title":"WaveGlow: A Flow-based Generative Network for Speech Synthesis. CoRR abs\/1811.00002","author":"Prenger Ryan","year":"2018","unstructured":"Ryan Prenger, Rafael Valle, and Bryan Catanzaro. 2018. WaveGlow: A Flow-based Generative Network for Speech Synthesis. CoRR abs\/1811.00002 (2018), 5\u00a0pages. http:\/\/arxiv.org\/abs\/1811.00002"},{"key":"e_1_3_2_1_13_1","doi-asserted-by":"publisher","unstructured":"Tim Sainburg. 2019. timsainb\/noisereduce v1.0. https:\/\/doi.org\/10.5281\/zenodo.3243139","DOI":"10.5281\/zenodo.3243139"},{"key":"e_1_3_2_1_14_1","volume-title":"Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions. CoRR abs\/1712.05884","author":"Shen Jonathan","year":"2017","unstructured":"Jonathan Shen, Ruoming Pang, Ron\u00a0J. Weiss, Mike Schuster, Navdeep Jaitly, Zongheng Yang, Zhifeng Chen, Yu Zhang, Yuxuan Wang, R.\u00a0J. Skerry-Ryan, Rif\u00a0A. Saurous, Yannis Agiomyrgiannakis, and Yonghui Wu. 2017. Natural TTS Synthesis by Conditioning WaveNet on Mel Spectrogram Predictions. CoRR abs\/1712.05884 (2017), 5\u00a0pages. http:\/\/arxiv.org\/abs\/1712.05884"},{"key":"e_1_3_2_1_15_1","unstructured":"TTSFree.com. 2023. TTSFree \u2013 Free Text-To-Speech and Text-to-MP3 for French. https:\/\/ttsfree.com\/text-to-speech\/french\/"},{"key":"e_1_3_2_1_16_1","unstructured":"TTSFree.com. 2023. TTSMP3 \u2013 Free Text-To-Speech and Text-to-MP3 for French. https:\/\/ttsmp3.com\/text-to-speech\/French\/"},{"key":"e_1_3_2_1_17_1","volume-title":"End-to-end Text-to-speech for Low-resource Languages by Cross-Lingual Transfer Learning. CoRR abs\/1904.06508","author":"Tu Tao","year":"2019","unstructured":"Tao Tu, Yuan-Jui Chen, Cheng-chieh Yeh, and Hung-yi Lee. 2019. End-to-end Text-to-speech for Low-resource Languages by Cross-Lingual Transfer Learning. CoRR abs\/1904.06508 (2019), 5\u00a0pages. http:\/\/arxiv.org\/abs\/1904.06508"},{"key":"e_1_3_2_1_18_1","volume-title":"WaveNet: A Generative Model for Raw Audio. CoRR abs\/1609.03499","author":"van\u00a0den Oord A\u00e4ron","year":"2016","unstructured":"A\u00e4ron van\u00a0den Oord, Sander Dieleman, Heiga Zen, Karen Simonyan, Oriol Vinyals, Alex Graves, Nal Kalchbrenner, Andrew Senior, and Koray Kavukcuoglu. 2016. WaveNet: A Generative Model for Raw Audio. CoRR abs\/1609.03499 (2016), 15\u00a0pages. http:\/\/arxiv.org\/abs\/1609.03499"},{"key":"e_1_3_2_1_19_1","doi-asserted-by":"publisher","unstructured":"Li Wan Quan Wang Alan Papir and Ignacio\u00a0Lopez Moreno. 2017. Generalized End-to-End Loss for Speaker Verification. https:\/\/doi.org\/10.48550\/ARXIV.1710.10467","DOI":"10.48550\/ARXIV.1710.10467"},{"key":"e_1_3_2_1_20_1","doi-asserted-by":"publisher","unstructured":"Chengyi Wang Sanyuan Chen Yu Wu Ziqiang Zhang Long Zhou Shujie Liu Zhuo Chen Yanqing Liu Huaming Wang Jinyu Li Lei He Sheng Zhao and Furu Wei. 2023. Neural Codec Language Models are Zero-Shot Text to Speech Synthesizers. https:\/\/doi.org\/10.48550\/ARXIV.2301.02111","DOI":"10.48550\/ARXIV.2301.02111"},{"key":"e_1_3_2_1_21_1","doi-asserted-by":"publisher","unstructured":"Junichi Yamagishi Pierre-Edouard Honnet Philip Garner and Alexandros Lazaridis. 2017. The SIWIS French Speech Synthesis Database. https:\/\/doi.org\/10.7488\/DS\/1705","DOI":"10.7488\/DS"},{"key":"e_1_3_2_1_22_1","volume-title":"A Comprehensive Survey on Transfer Learning. CoRR abs\/1911.02685","author":"Zhuang Fuzhen","year":"2019","unstructured":"Fuzhen Zhuang, Zhiyuan Qi, Keyu Duan, Dongbo Xi, Yongchun Zhu, Hengshu Zhu, Hui Xiong, and Qing He. 2019. A Comprehensive Survey on Transfer Learning. CoRR abs\/1911.02685 (2019), 31\u00a0pages. http:\/\/arxiv.org\/abs\/1911.02685"}],"event":{"name":"PETRA '23: Proceedings of the 16th International Conference on PErvasive Technologies Related to Assistive Environments","acronym":"PETRA '23","location":"Corfu Greece"},"container-title":["Proceedings of the 16th International Conference on PErvasive Technologies Related to Assistive Environments"],"original-title":[],"link":[{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3594806.3596536","content-type":"unspecified","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/dl.acm.org\/doi\/pdf\/10.1145\/3594806.3596536","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,6,17]],"date-time":"2025-06-17T17:49:01Z","timestamp":1750182541000},"score":1,"resource":{"primary":{"URL":"https:\/\/dl.acm.org\/doi\/10.1145\/3594806.3596536"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2023,7,5]]},"references-count":22,"alternative-id":["10.1145\/3594806.3596536","10.1145\/3594806"],"URL":"https:\/\/doi.org\/10.1145\/3594806.3596536","relation":{},"subject":[],"published":{"date-parts":[[2023,7,5]]},"assertion":[{"value":"2023-08-10","order":3,"name":"published","label":"Published","group":{"name":"publication_history","label":"Publication History"}}]}}