{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:21Z","timestamp":1779228381275,"version":"3.51.4"},"reference-count":125,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,4,3]],"date-time":"2026-04-03T00:00:00Z","timestamp":1775174400000},"content-version":"vor","delay-in-days":0,"URL":"http:\/\/creativecommons.org\/licenses\/by-nc\/4.0\/"}],"funder":[{"DOI":"10.13039\/100016308","name":"Banque Publique d'Investissement France","doi-asserted-by":"publisher","id":[{"id":"10.13039\/100016308","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100010190","name":"Grand \u00c9quipement National De Calcul Intensif","doi-asserted-by":"publisher","id":[{"id":"10.13039\/501100010190","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101985","type":"journal-article","created":{"date-parts":[[2026,3,26]],"date-time":"2026-03-26T16:39:04Z","timestamp":1774543144000},"page":"101985","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["A closer look at internal representations of end-to-end Text-to-Speech models: How is phonetic and acoustic information encoded?"],"prefix":"10.1016","volume":"100","author":[{"ORCID":"https:\/\/orcid.org\/0000-0002-3359-6846","authenticated-orcid":false,"given":"Martin","family":"Lenglet","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-9909-6078","authenticated-orcid":false,"given":"Olivier","family":"Perrotin","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6053-0818","authenticated-orcid":false,"given":"G\u00e9rard","family":"Bailly","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101985_b1","doi-asserted-by":"crossref","unstructured":"Adigwe, Adaeze, King, Simon, Lai, Catherine, 2025. Modelling degrees of Spontaneity in Text-to-Speech Synthesis. In: ISCA Speech Synthesis Workshop. Leeuwarden, The Netherlands, pp. 96\u2013103. http:\/\/dx.doi.org\/10.21437\/SSW.2025-15.","DOI":"10.21437\/SSW.2025-15"},{"key":"10.1016\/j.csl.2026.101985_b2","series-title":"AAAI Conference on Artificial Intelligence","first-page":"3159","article-title":"Character-level language modeling with deeper self-attention","volume":"Vol. 33","author":"Al-Rfou","year":"2019"},{"key":"10.1016\/j.csl.2026.101985_b3","unstructured":"Baevski, Alexei, Zhou, Yuhao, Mohamed, Abdelrahman, Auli, Michael, 2020. wav2vec 2.0: A Framework for Self-Supervised Learning of Speech Representations. In: Advances in Neural Information Processing Systems. Vol. 33, Virtual, pp. 12449\u201312460, URL https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/file\/92d1e1eb1cd6f9fba3227870bb6d7f07-Paper.pdf."},{"key":"10.1016\/j.csl.2026.101985_b4","doi-asserted-by":"crossref","unstructured":"Bailly, G\u00e9rard, Andr\u00e9, Elisabeth, Cooper, Erica, Klabbers, Esther, Cowan, Benjamin, Edlund, Jens, Harte, Naomi, King, Simon, Le Maguer, S\u00e9bastien, Moore, Roger K., M\u00f6bius, Bernd, M\u00f6ller, Sebastian, Pandey, Ayushi, Perrotin, Olivier, Seebauer, Fritz, Str\u00f6mbergsson, Sofia, Traum, David R., T\u00e5nnander, Christina, Wagner, Petra, Yamagishi, Junichi, Yasuda, Yusuke, 2025. Hot topics in speech synthesis evaluation. In: ISCA Speech Synthesis Workshop. Leeuwarden, The Netherlands, pp. 1\u20137. http:\/\/dx.doi.org\/10.21437\/SSW.2025-1.","DOI":"10.21437\/SSW.2025-1"},{"key":"10.1016\/j.csl.2026.101985_b5","doi-asserted-by":"crossref","unstructured":"Bailly, G\u00e9rard, Legrand, Romain, Lenglet, Martin, Elisei, Fr\u00e9d\u00e9ric, Garnier, Ma\u00ebva, Perrotin, Olivier, 2024. Emotags: Computer-Assisted Verbal Labelling of Expressive Audiovisual Utterances for Expressive Multimodal TTS. In: Joint International Conference on Computational Linguistics, Language Resources and Evaluation. LREC-COLING, Torino, Italia, URL https:\/\/aclanthology.org\/2024.lrec-main.505.","DOI":"10.63317\/2c6ik5stbyej"},{"key":"10.1016\/j.csl.2026.101985_b6","doi-asserted-by":"crossref","unstructured":"Bailly, G\u00e9rard, Lenglet, Martin, Perrotin, Olivier, Klabbers, Esther, 2023. Advocating for text input in multi-speaker text-to-speech systems. In: ISCA Speech Synthesis Workshop . SSW, Grenoble, France, pp. 1\u20137. http:\/\/dx.doi.org\/10.21437\/SSW.2023-1.","DOI":"10.21437\/SSW.2023-1"},{"key":"10.1016\/j.csl.2026.101985_b7","doi-asserted-by":"crossref","unstructured":"Bailly, G\u00e9rard, Perrotin, Olivier, 2025. The GIPSA-Lab Text-To-Speech System for the Blizzard Challenge 2025. In: Blizzard Challenge Workshop. Leeuwarden, The Netherlands, URL https:\/\/www.isca-archive.org\/blizzard_2025\/bailly25_blizzard.pdf.","DOI":"10.21437\/Blizzard.2025-5"},{"key":"10.1016\/j.csl.2026.101985_b8","series-title":"Ressources for end-to-end french text-to-speech blizzard challenge","author":"Bailly","year":"2021"},{"key":"10.1016\/j.csl.2026.101985_b9","unstructured":"Bau, Anthony, Belinkov, Yonatan, Sajjad, Hassan, Durrani, Nadir, Dalvi, Fahim, Glass, James, 2019. Identifying and Controlling Important Neurons in Neural Machine Translation. In: International Conference on Learning Representations. ICLR, New Orleans, LA, USA, URL https:\/\/openreview.net\/forum?id=H1z-PsR5KX."},{"key":"10.1016\/j.csl.2026.101985_b10","doi-asserted-by":"crossref","first-page":"43","DOI":"10.1016\/j.langcom.2013.12.007","article-title":"Linguistic repertoire and ethnic identity in New York City","volume":"35","author":"Becker","year":"2014","journal-title":"Lang. Commun."},{"issue":"1","key":"10.1016\/j.csl.2026.101985_b11","doi-asserted-by":"crossref","first-page":"207","DOI":"10.1162\/coli_a_00422","article-title":"Probing classifiers: Promises, shortcomings, and advances","volume":"48","author":"Belinkov","year":"2022","journal-title":"Comput. Linguist."},{"issue":"9","key":"10.1016\/j.csl.2026.101985_b12","first-page":"341","article-title":"Praat, a system for doing phonetics by computer","volume":"5","author":"Boersma","year":"2001","journal-title":"Glot Int."},{"key":"10.1016\/j.csl.2026.101985_b13","series-title":"Intonation And Its Uses: Melody In Grammar and Discourse","author":"Bolinger","year":"1989"},{"key":"10.1016\/j.csl.2026.101985_b14","series-title":"AudioLM: a language modeling approach to audio generation","author":"Borsos","year":"2023"},{"issue":"2","key":"10.1016\/j.csl.2026.101985_b15","doi-asserted-by":"crossref","first-page":"230","DOI":"10.1111\/j.1467-9817.2008.01387.x","article-title":"Influence of the visual attention span on child reading performance: a cross-sectional study","volume":"32","author":"Bosse","year":"2009","journal-title":"J. Res. Read."},{"key":"10.1016\/j.csl.2026.101985_b16","doi-asserted-by":"crossref","first-page":"245","DOI":"10.1613\/jair.1.12228","article-title":"A survey on the explainability of supervised machine learning","volume":"70","author":"Burkart","year":"2021","journal-title":"J. Artificial Intelligence Res."},{"key":"10.1016\/j.csl.2026.101985_b17","doi-asserted-by":"crossref","unstructured":"Chen, Li-Wei, Rudnicky, Alexander, 2022. Fine-grained style control in Transformer-based Text-to-speech Synthesis. In: IEEE International Conference on Acoustics, Speech and Signal Processing . ICASSP, Singapore, Singapore, pp. 7907\u20137911. http:\/\/dx.doi.org\/10.1109\/ICASSP43922.2022.9747747.","DOI":"10.1109\/ICASSP43922.2022.9747747"},{"key":"10.1016\/j.csl.2026.101985_b18","doi-asserted-by":"crossref","first-page":"705","DOI":"10.1109\/TASLPRO.2025.3530270","article-title":"Neural codec language models are zero-shot text to speech synthesizers","volume":"33","author":"Chen","year":"2025","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101985_b19","doi-asserted-by":"crossref","unstructured":"Chien, Chung-Ming, Lin, Jheng-Hao, Huang, Chien-yu, Hsu, Po-chun, Lee, Hung-yi, 2021. Investigating on Incorporating Pretrained and Learnable Speaker Representations for Multi-Speaker Multi-Style Text-to-Speech. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, Toronto, Canada, pp. 8588\u20138592. http:\/\/dx.doi.org\/10.1109\/ICASSP39728.2021.9413880.","DOI":"10.1109\/ICASSP39728.2021.9413880"},{"key":"10.1016\/j.csl.2026.101985_b20","unstructured":"Chorowski, Jan K, Bahdanau, Dzmitry, Serdyuk, Dmitriy, Cho, Kyunghyun, Bengio, Yoshua, 2015. Attention-based models for speech recognition. In: Advances in Neural Information Processing Systems. Montreal, Canada, pp. 577\u2013585, URL https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2015\/file\/1068c6e4c8051cfd4e9ea8072e3189e2-Paper.pdf."},{"key":"10.1016\/j.csl.2026.101985_b21","series-title":"Intonation","author":"Cruttenden","year":"1997"},{"key":"10.1016\/j.csl.2026.101985_b22","article-title":"High fidelity neural audio compression","author":"D\u00e9fossez","year":"2023","journal-title":"Trans. Mach. Learn. Res."},{"key":"10.1016\/j.csl.2026.101985_b23","series-title":"Moshi: a speech-text foundation model for real-time dialogue","author":"D\u00e9fossez","year":"2024"},{"key":"10.1016\/j.csl.2026.101985_b24","series-title":"Proc. Conference of the North American Chapter of the Association for Computational Linguistics: Human Language Technologies","first-page":"4171","article-title":"BERT: Pre-training of deep bidirectional transformers for language understanding","author":"Devlin","year":"2019"},{"key":"10.1016\/j.csl.2026.101985_b25","series-title":"CosyVoice: A scalable multilingual zero-shot text-to-speech synthesizer based on supervised semantic tokens","author":"Du","year":"2024"},{"key":"10.1016\/j.csl.2026.101985_b26","doi-asserted-by":"crossref","unstructured":"Eskimez, Sefik Emre, Wang, Xiaofei, Thakker, Manthan, Li, Canrun, Tsai, Chung-Hsien, Xiao, Zhen, Yang, Hemin, Zhu, Zirun, Tang, Min, Tan, Xu, Liu, Yanqing, Zhao, Sheng, Kanda, Naoyuki, 2024. E2 TTS: Embarrassingly Easy Fully Non-Autoregressive Zero-Shot TTS. In: IEEE Spoken Language Technology Workshop. SLT, Macao, pp. 682\u2013689. http:\/\/dx.doi.org\/10.1109\/SLT61566.2024.10832320.","DOI":"10.1109\/SLT61566.2024.10832320"},{"issue":"2","key":"10.1016\/j.csl.2026.101985_b27","doi-asserted-by":"crossref","first-page":"190","DOI":"10.1109\/TAFFC.2015.2457417","article-title":"The geneva minimalistic acoustic parameter set (GeMAPS) for voice research and affective computing","volume":"7","author":"Eyben","year":"2015","journal-title":"IEEE Trans. Affect. Comput."},{"issue":"4","key":"10.1016\/j.csl.2026.101985_b28","doi-asserted-by":"crossref","first-page":"409","DOI":"10.1016\/j.wocn.2005.08.002","article-title":"The social life of phonetics and phonology","volume":"34","author":"Foulkes","year":"2006","journal-title":"J. Phon."},{"key":"10.1016\/j.csl.2026.101985_b29","doi-asserted-by":"crossref","first-page":"74","DOI":"10.1016\/j.csl.2017.07.003","article-title":"Audio\u2013visual synchronization in reading while listening to texts: Effects on visual behavior and verbal learning","volume":"47","author":"Gerbier","year":"2018","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.csl.2026.101985_b30","doi-asserted-by":"crossref","unstructured":"Godde, Erika, Bailly, G\u00e9rard, Escudero, David, Bosse, Marie-Line, Bianco, Maryse, Vilain, Coriandre Emmanuel, 2017. Improving fluency of young readers: introducing a Karaoke to learn how to breath during a Reading-while-Listening task. In: ISCA Workshop on Speech and Language Technology in Education SLaTE. Stockholm, Sweden, pp. 127\u2013131. http:\/\/dx.doi.org\/10.21437\/SLaTE.2017-22.","DOI":"10.21437\/SLaTE.2017-22"},{"key":"10.1016\/j.csl.2026.101985_b31","doi-asserted-by":"crossref","first-page":"4036","DOI":"10.1109\/TASLP.2024.3451951","article-title":"ZMM-TTS: Zero-shot multilingual and multispeaker speech synthesis conditioned on self-supervised discrete speech representations","volume":"32","author":"Gong","year":"2024","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"issue":"4","key":"10.1016\/j.csl.2026.101985_b32","doi-asserted-by":"crossref","DOI":"10.1002\/ail2.61","article-title":"DARPA\u2019s explainable AI (XAI) program: A retrospective","volume":"2","author":"Gunning","year":"2021","journal-title":"Appl. AI Lett."},{"key":"10.1016\/j.csl.2026.101985_b33","series-title":"Improving neural networks by preventing co-adaptation of feature detectors","author":"Hinton","year":"2012"},{"issue":"8","key":"10.1016\/j.csl.2026.101985_b34","doi-asserted-by":"crossref","first-page":"1735","DOI":"10.1162\/neco.1997.9.8.1735","article-title":"Long short-term memory","volume":"9","author":"Hochreiter","year":"1997","journal-title":"Neural Comput."},{"key":"10.1016\/j.csl.2026.101985_b35","doi-asserted-by":"crossref","first-page":"3451","DOI":"10.1109\/TASLP.2021.3122291","article-title":"HuBERT: Self-supervised speech representation learning by masked prediction of hidden units","volume":"29","author":"Hsu","year":"2021","journal-title":"IEEE\/ACM Trans. Audio Speech Lang. Process."},{"key":"10.1016\/j.csl.2026.101985_b36","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"5901","article-title":"Disentangling correlated speaker and noise for speech synthesis via data augmentation and adversarial factorization","author":"Hsu","year":"2019"},{"key":"10.1016\/j.csl.2026.101985_b37","doi-asserted-by":"crossref","unstructured":"Hu, Tianyang, Chen, Fei, Wang, Haonan, Li, Jiawei, Wang, Wenjia, Sun, Jiacheng, Li, Zhenguo, 2023. Complexity matters: Rethinking the latent space for generative modeling. In: Advances in Neural Information Processing Systems. Vol. 36, New Orleans, LA, USA, pp. 29558\u201329579, URL https:\/\/openreview.net\/forum?id=00EKYYu3fD.","DOI":"10.52202\/075280-1285"},{"key":"10.1016\/j.csl.2026.101985_b38","series-title":"IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"11916","article-title":"Controllable prosody generation with partial inputs","author":"Iliescu","year":"2024"},{"key":"10.1016\/j.csl.2026.101985_b39","series-title":"IEEE International Conference on Acoustics, Speech, and Signal Processing Workshops","article-title":"Exploring the multidimensional representation of unidimensional speech acoustic parameters extracted by deep unsupervised models","author":"Jacquelin","year":"2024"},{"key":"10.1016\/j.csl.2026.101985_b40","series-title":"Diff-tts: A denoising diffusion model for text-to-speech","author":"Jeong","year":"2021"},{"key":"10.1016\/j.csl.2026.101985_b41","unstructured":"Jia, Ye, Zhang, Yu, Weiss, Ron, Wang, Quan, Shen, Jonathan, Ren, Fei, Chen, zhifeng, Nguyen, Patrick, Pang, Ruoming, Lopez Moreno, Ignacio, Wu, Yonghui, 2018. Transfer Learning from Speaker Verification to Multispeaker Text-To-Speech Synthesis. In: Advances in Neural Information Processing Systems. Vol. 31, Montreal, Canada, URL https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2018\/file\/6832a7b24bc06775d02b7406880b93fc-Paper.pdf."},{"key":"10.1016\/j.csl.2026.101985_b42","series-title":"Annual Meeting of the Association for Computational Linguistics","first-page":"655","article-title":"A convolutional neural network for modelling sentences","author":"Kalchbrenner","year":"2014"},{"issue":"2","key":"10.1016\/j.csl.2026.101985_b43","doi-asserted-by":"crossref","first-page":"27","DOI":"10.1007\/s13347-023-00606-x","article-title":"In conversation with artificial intelligence: aligning language models with human values","volume":"36","author":"Kasirzadeh","year":"2023","journal-title":"Philos. Technol."},{"key":"10.1016\/j.csl.2026.101985_b44","doi-asserted-by":"crossref","unstructured":"Kastner, Kyle, Santos, Jo\u00e3o Felipe, Bengio, Yoshua, Courville, Aaron, 2019. Representation mixing for TTS synthesis. In: IEEE International Conference on Acoustics, Speech and Signal Processing . ICASSP, Brighton, UK, pp. 5906\u20135910. http:\/\/dx.doi.org\/10.1109\/ICASSP.2019.8682880.","DOI":"10.1109\/ICASSP.2019.8682880"},{"key":"10.1016\/j.csl.2026.101985_b45","doi-asserted-by":"crossref","unstructured":"Kim, Minchan, Cheon, Sung Jun, Choi, Byoung Jin, Kim, Jong Jin, Kim, Nam Soo, 2021. Expressive Text-to-Speech Using Style Tag. In: Interspeech. Brno, Czechia, pp. 4663\u20134667. http:\/\/dx.doi.org\/10.21437\/Interspeech.2021-465.","DOI":"10.21437\/Interspeech.2021-465"},{"key":"10.1016\/j.csl.2026.101985_b46","unstructured":"Kim, Jaehyeon, Kim, Sungwon, Kong, Jungil, Yoon, Sungroh, 2020. Glow-TTS: a generative flow for text-to-speech via monotonic alignment search. In: Advances in Neural Information Processing Systems. Vancouver, BC, Canada, pp. 8067\u20138077, URL https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2020\/hash\/5c3b99e8f92532e5ad1556e53ceea00c-Abstract.html."},{"key":"10.1016\/j.csl.2026.101985_b47","unstructured":"Kim, Jaehyeon, Kong, Jungil, Son, Juhee, 2021. Conditional Variational Autoencoder with Adversarial Learning for End-to-End Text-to-Speech. In: Proceedings of the 38th International Conference on Machine Learning. In: Proceedings of Machine Learning Research, Vol. 139, Virtual, pp. 5530\u20135540, URL https:\/\/proceedings.mlr.press\/v139\/kim21f.html."},{"key":"10.1016\/j.csl.2026.101985_b48","unstructured":"Kong, Jungil, Kim, Jaehyeon, Bae, Jaekyoung, 2020. HiFi-GAN: Generative adversarial networks for efficient and high fidelity speech synthesis. In: Advances in Neural Information Processing Systems. Vol. 33, Vancouver, Canada, pp. 17022\u201317033, URL https:\/\/proceedings.neurips.cc\/paper\/2020\/hash\/c5d736809766d46260d816d8dbc9eb44-Abstract.html."},{"key":"10.1016\/j.csl.2026.101985_b49","series-title":"Multidimensional Scaling","author":"Kruskal","year":"1978"},{"key":"10.1016\/j.csl.2026.101985_b50","doi-asserted-by":"crossref","first-page":"1336","DOI":"10.1162\/tacl_a_00430","article-title":"On generative spoken language modeling from raw audio","volume":"9","author":"Lakhotia","year":"2021","journal-title":"Trans. Assoc. Comput. Linguist."},{"key":"10.1016\/j.csl.2026.101985_b51","doi-asserted-by":"crossref","unstructured":"Lameris, Harm, Gustafsson, Joakim, Sz\u00e9kely, \u00c9va, 2025. VoiceQualityVC: A Voice Conversion System for Studying the Perceptual Effects of Voice Quality in Speech. In: Interspeech. Rotterdam, The Netherlands, pp. 2295\u20132299. http:\/\/dx.doi.org\/10.21437\/Interspeech.2025-902.","DOI":"10.21437\/Interspeech.2025-902"},{"key":"10.1016\/j.csl.2026.101985_b52","doi-asserted-by":"crossref","unstructured":"Lameris, Harm, Ward, Nigel, 2025. Creakiness, Breathiness, and Nasality Contribute to the Perceived Suitability of Synthesized Speech in a Pragmatically-Rich Domain. In: ISCA Speech Synthesis Workshop. Leeuwarden, The Netherlands, pp. 89\u201395. http:\/\/dx.doi.org\/10.21437\/SSW.2025-14.","DOI":"10.21437\/SSW.2025-14"},{"key":"10.1016\/j.csl.2026.101985_b53","series-title":"Empirical Methods in Natural Language Processing","article-title":"From perception to production: how acoustic invariance facilitates articulatory learning in a self-supervised vocal imitation model","author":"Lavechin","year":"2025"},{"key":"10.1016\/j.csl.2026.101985_b54","doi-asserted-by":"crossref","unstructured":"Lee, Younggun, Kim, Taesu, 2019. Robust and Fine-grained Prosody Control of End-to-end Speech Synthesis. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, Brighton, UK, pp. 5911\u20135915. http:\/\/dx.doi.org\/10.1109\/ICASSP.2019.8683501.","DOI":"10.1109\/ICASSP.2019.8683501"},{"key":"10.1016\/j.csl.2026.101985_b55","unstructured":"Lee, Sang-gil, Ping, Wei, Ginsburg, Boris, Catanzaro, Bryan, Yoon, Sungroh, 2023. BigVGAN: A Universal Neural Vocoder with Large-Scale Training. In: International Conference on Learning Representations. ICLR, Kigali, Rwanda, URL https:\/\/openreview.net\/forum?id=iTtGCMDEzS_."},{"key":"10.1016\/j.csl.2026.101985_b56","doi-asserted-by":"crossref","unstructured":"Lemerle, Th\u00e9odor, Obin, Nicolas, Roebel, Axel, 2025. Lina-Style: Word-Level Style Control in TTS via Interleaved Synthetic Data. In: ISCA Speech Synthesis Workshop. Leeuwarden, The Netherlands, pp. 35\u201339. http:\/\/dx.doi.org\/10.21437\/SSW.2025-6.","DOI":"10.21437\/SSW.2025-6"},{"key":"10.1016\/j.csl.2026.101985_b57","series-title":"Analysis of Latent Representations of Neural Text-To-Speech Models for Expressive Audio-Visual Synthesis","author":"Lenglet","year":"2023"},{"key":"10.1016\/j.csl.2026.101985_b58","doi-asserted-by":"crossref","unstructured":"Lenglet, Martin, Perrotin, Olivier, Bailly, G\u00e9rard, 2021. Impact of Segmentation and Annotation in French end-to-end Synthesis. In: ISCA Speech Synthesis Workshop. SSW, Budapest, Hungary, pp. 13\u201318. http:\/\/dx.doi.org\/10.21437\/SSW.2021-3.","DOI":"10.21437\/SSW.2021-3"},{"key":"10.1016\/j.csl.2026.101985_b59","doi-asserted-by":"crossref","unstructured":"Lenglet, Martin, Perrotin, Olivier, Bailly, G\u00e9rard, 2022a. Mod\u00e9lisation de la Parole avec Tacotron2 : Analyse acoustique et phon\u00e9tique des plongements de caract\u00e8re. In: Actes des Journ\u00e9es d\u2019Etudes sur la Parole. JEP, Noirmoutiers, France, pp. 788\u2013796. http:\/\/dx.doi.org\/10.21437\/JEP.2022-83.","DOI":"10.21437\/JEP.2022-83"},{"key":"10.1016\/j.csl.2026.101985_b60","doi-asserted-by":"crossref","unstructured":"Lenglet, Martin, Perrotin, Olivier, Bailly, G\u00e9rard, 2022b. Speaking Rate Control of end-to-end TTS Models by Direct Manipulation of the Encoder\u2019s Output Embeddings. In: Interspeech. Incheon, Korea, pp. 11\u201315. http:\/\/dx.doi.org\/10.21437\/Interspeech.2022-759.","DOI":"10.21437\/Interspeech.2022-759"},{"key":"10.1016\/j.csl.2026.101985_b61","doi-asserted-by":"crossref","unstructured":"Lenglet, Martin, Perrotin, Olivier, Bailly, G\u00e9rard, 2023a. The GIPSA-Lab Text-To-Speech System for the Blizzard Challenge 2023. In: Blizzard Challenge Workshop. Grenoble, France, pp. 34\u201339, URL https:\/\/www.isca-archive.org\/blizzard_2023\/lenglet23_blizzard.pdf.","DOI":"10.21437\/Blizzard.2023-3"},{"key":"10.1016\/j.csl.2026.101985_b62","doi-asserted-by":"crossref","unstructured":"Lenglet, Martin, Perrotin, Olivier, Bailly, G\u00e9rard, 2023b. Local Style Tokens: Fine-Grained Prosodic Representations For TTS Expressive Control. In: ISCA Speech Synthesis Workshop. SSW, Grenoble, France, pp. 120\u2013126. http:\/\/dx.doi.org\/10.21437\/SSW.2023-19.","DOI":"10.21437\/SSW.2023-19"},{"key":"10.1016\/j.csl.2026.101985_b63","doi-asserted-by":"crossref","unstructured":"Li, Yinghao Aaron, Han, Cong, Raghavan, Vinay, Mischler, Gavin, Mesgarani, Nima, 2023. StyleTTS 2: Towards Human-Level Text-to-Speech through Style Diffusion and Adversarial Training with Large Speech Language Models. In: Advances in Neural Information Processing Systems. Vol. 36, New Orleans, LA, USA, pp. 19594\u201319621, URL https:\/\/proceedings.neurips.cc\/paper_files\/paper\/2023\/file\/3eaad2a0b62b5ed7a2e66c2188bb1449-Paper-Conference.pdf.","DOI":"10.52202\/075280-0860"},{"issue":"1","key":"10.1016\/j.csl.2026.101985_b64","doi-asserted-by":"crossref","first-page":"411","DOI":"10.1121\/1.428140","article-title":"Effect of vocal effort on spectral properties of vowels","volume":"106","author":"Li\u00e9nard","year":"1999","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101985_b65","series-title":"Interspeech","first-page":"21","article-title":"JETS: Jointly training FastSpeech2 and HiFi-GAN for end to end text to speech","author":"Lim","year":"2022"},{"key":"10.1016\/j.csl.2026.101985_b66","unstructured":"Lin, Weiwei, He, Chenghan, 2025. Continuous Autoregressive Modeling with Stochastic Monotonic Alignment for Speech Synthesis. In: International Conference on Learning Representations. ICLR, Singapore, URL https:\/\/openreview.net\/forum?id=cuFzE8Jlvb."},{"key":"10.1016\/j.csl.2026.101985_b67","doi-asserted-by":"crossref","first-page":"902","DOI":"10.1038\/s41593-025-01905-6","article-title":"A streaming brain-to-voice neuroprosthesis to restore naturalistic communication","volume":"28","author":"Littlejohn","year":"2025","journal-title":"Nature Neurosci."},{"key":"10.1016\/j.csl.2026.101985_b68","doi-asserted-by":"crossref","unstructured":"Liu, Zhijun, Guo, Yiwei, Yu, Kai, 2023. DiffVoice: Text-to-Speech with Latent Diffusion. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, Rhodes Island, Greece, pp. 1\u20135. http:\/\/dx.doi.org\/10.1109\/ICASSP49357.2023.10095100.","DOI":"10.1109\/ICASSP49357.2023.10095100"},{"key":"10.1016\/j.csl.2026.101985_b69","doi-asserted-by":"crossref","unstructured":"Lux, Florian, Meyer, Sarina, Behringer, Lyonel, Zalkow, Frank, Do, Phat, Coler, Matt, Habets, Emanu\u00ebl A. P., Vu, Ngoc Thang, 2024. Meta Learning Text-to-Speech Synthesis in over 7000 Languages. In: Interspeech. Kos, Greece, pp. 4958\u20134962. http:\/\/dx.doi.org\/10.21437\/Interspeech.2024-1335.","DOI":"10.21437\/Interspeech.2024-1335"},{"key":"10.1016\/j.csl.2026.101985_b70","series-title":"Natural language guidance of high-fidelity text-to-speech with synthetic annotations","author":"Lyth","year":"2024"},{"key":"10.1016\/j.csl.2026.101985_b71","doi-asserted-by":"crossref","unstructured":"Miao, Chenfeng, Liang, Shuang, Chen, Minchuan, Ma, Jun, Wang, Shaojun, Xiao, Jing, 2020. Flow-TTS: A Non-Autoregressive Network for Text to Speech Based on Flow. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, Barcelona, Spain, pp. 7209\u20137213. http:\/\/dx.doi.org\/10.1109\/ICASSP40776.2020.9054484.","DOI":"10.1109\/ICASSP40776.2020.9054484"},{"key":"10.1016\/j.csl.2026.101985_b72","unstructured":"Min, Dongchan, Lee, Dong Bok, Yang, Eunho, Hwang, Sung Ju, 2021. Meta-stylespeech: Multi-speaker adaptive text-to-speech generation. In: International Conference on Machine Learning. ICML, Vol. 139, Virtual, pp. 7748\u20137759, URL https:\/\/proceedings.mlr.press\/v139\/min21b\/min21b.pdf."},{"issue":"3","key":"10.1016\/j.csl.2026.101985_b73","doi-asserted-by":"crossref","first-page":"291","DOI":"10.1016\/j.wocn.2003.11.002","article-title":"An acoustic analysis of the bidirectionality of coarticulation in VCV utterances","volume":"32","author":"Modarresi","year":"2004","journal-title":"J. Phon."},{"key":"10.1016\/j.csl.2026.101985_b74","doi-asserted-by":"crossref","unstructured":"Mohan, Devang S. Ram, Hu, Vivian, Teh, Tian Huey, Torresquintero, Alexandra, Wallis, Christopher G.R., Staib, Marlene, Foglianti, Lorenzo, Gao, Jiameng, King, Simon, 2021. Ctrl-P: Temporal Control of Prosodic Variation for Speech Synthesis. In: Interspeech. Brno, Czechia, pp. 3875\u20133879. http:\/\/dx.doi.org\/10.21437\/Interspeech.2021-1583.","DOI":"10.21437\/Interspeech.2021-1583"},{"key":"10.1016\/j.csl.2026.101985_b75","doi-asserted-by":"crossref","unstructured":"Montero, Milton, Bowers, Jeffrey, Ponte Costa, Rui, Ludwig, Casimir, Malhotra, Gaurav, 2022. Lost in latent space: Examining failures of disentangled models at combinatorial generalisation. In: Advances in Neural Information Processing Systems. Vol. 35, New Orleans, LA, USA, pp. 10136\u201310149, URL https:\/\/openreview.net\/forum?id=7yUxTNWyQGf.","DOI":"10.52202\/068431-0736"},{"issue":"67\u201368","key":"10.1016\/j.csl.2026.101985_b76","first-page":"7","article-title":"Numerical optimization","volume":"35","author":"Nocedal","year":"1999","journal-title":"Springer Sci."},{"key":"10.1016\/j.csl.2026.101985_b77","series-title":"VibeVoice technical report","author":"Peng","year":"2025"},{"key":"10.1016\/j.csl.2026.101985_b78","series-title":"An investigation of the relation between grapheme embeddings and pronunciation for tacotron-based systems","author":"Perquin","year":"2021"},{"key":"10.1016\/j.csl.2026.101985_b79","doi-asserted-by":"crossref","unstructured":"Perrotin, Olivier, Amouri, Hussein El, Bailly, G\u00e9rard, Hueber, Thomas, 2021. Evaluating the Extrapolation Capabilities of Neural Vocoders to Extreme Pitch Values. In: Interspeech. Brno, Czechia, pp. 11\u201315. http:\/\/dx.doi.org\/10.21437\/Interspeech.2021-1547.","DOI":"10.21437\/Interspeech.2021-1547"},{"key":"10.1016\/j.csl.2026.101985_b80","doi-asserted-by":"crossref","DOI":"10.1016\/j.csl.2024.101747","article-title":"Refining the evaluation of speech synthesis: A summary of the Blizzard challenge 2023","volume":"90","author":"Perrotin","year":"2025","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.csl.2026.101985_b81","unstructured":"Popov, Vadim, Vovk, Ivan, Gogoryan, Vladimir, Sadekova, Tasnima, Kudinov, Mikhail, 2021. Grad-TTS: A Diffusion Probabilistic Model for Text-to-Speech. In: International Conference on Machine Learning. ICML, Vol. 139, Virtual, pp. 8599\u20138608, URL https:\/\/proceedings.mlr.press\/v139\/popov21a.html."},{"key":"10.1016\/j.csl.2026.101985_b82","doi-asserted-by":"crossref","unstructured":"Prenger, Ryan, Valle, Rafael, Catanzaro, Bryan, 2019. Waveglow: A flow-based generative network for speech synthesis. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, Brighton, UK, pp. 3617\u20133621. http:\/\/dx.doi.org\/10.1109\/ICASSP.2019.8683143.","DOI":"10.1109\/ICASSP.2019.8683143"},{"key":"10.1016\/j.csl.2026.101985_b83","series-title":"Improving Language Understanding by Generative Pre-Training","author":"Radford","year":"2018"},{"key":"10.1016\/j.csl.2026.101985_b84","doi-asserted-by":"crossref","unstructured":"Raitio, Tuomo, Rasipuram, Ramya, Castellani, Dan, 2020. Controllable Neural Text-to-Speech Synthesis Using Intuitive Prosodic Features. In: Interspeech. Shanghai, China, pp. 4432\u20134436. http:\/\/dx.doi.org\/10.21437\/Interspeech.2020-2861.","DOI":"10.21437\/Interspeech.2020-2861"},{"key":"10.1016\/j.csl.2026.101985_b85","doi-asserted-by":"crossref","unstructured":"R\u00e4uker, Tilman, Ho, Anson, Casper, Stephen, Hadfield-Menell, Dylan, 2023. Toward transparent AI: A survey on interpreting the inner structures of deep neural networks. In: IEEE Conference on Secure and Trustworthy Machine Learning. SaTML, Raleigh, NC, USA, pp. 464\u2013483. http:\/\/dx.doi.org\/10.1109\/SaTML54575.2023.00039.","DOI":"10.1109\/SaTML54575.2023.00039"},{"key":"10.1016\/j.csl.2026.101985_b86","doi-asserted-by":"crossref","unstructured":"Rautenberg, Frederik, Kuhlmann, Michael, Seebauer, Fritz, Wiechmann, Jana, Wagner, Petra, Haeb-Umbach, Reinhold, 2025. Speech Synthesis along Perceptual Voice Quality Dimensions. In: IEEE International Conference on Acoustics, Speech, and Signal Processing. ICASSP, Hyderabad, India, pp. 1\u20135. http:\/\/dx.doi.org\/10.1109\/ICASSP49660.2025.10888012.","DOI":"10.1109\/ICASSP49660.2025.10888012"},{"key":"10.1016\/j.csl.2026.101985_b87","unstructured":"Ren, Yi, Hu, Chenxu, Tan, Xu, Qin, Tao, Zhao, Sheng, Zhao, Zhou, Liu, Tie-Yan, 2021. FastSpeech 2: Fast and High-Quality End-to-End Text to Speech. In: International Conference on Learning Representations. ICLR, Virtual, http:\/\/dx.doi.org\/10.48550\/ARXIV.2006.04558."},{"key":"10.1016\/j.csl.2026.101985_b88","series-title":"Phonological variants and dialect identification in Latin American Spanish","author":"Resnick","year":"2012"},{"key":"10.1016\/j.csl.2026.101985_b89","doi-asserted-by":"crossref","unstructured":"Sadok, Samir, Hauret, Julien, Bavu, \u00c9ric, 2025. Bringing Interpretability to Neural Audio Codecs. In: Interspeech. Rotterdam, The Netherlands, pp. 5023\u20135027. http:\/\/dx.doi.org\/10.21437\/Interspeech.2025-115.","DOI":"10.21437\/Interspeech.2025-115"},{"key":"10.1016\/j.csl.2026.101985_b90","doi-asserted-by":"crossref","first-page":"53","DOI":"10.1016\/j.specom.2023.02.005","article-title":"Learning and controlling the source-filter representation of speech with a variational autoencoder","volume":"148","author":"Sadok","year":"2023","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101985_b91","doi-asserted-by":"crossref","unstructured":"Sang, Dinh Viet, Thu, Lam Xuan, 2021. FastTacotron: A Fast, Robust and Controllable Method for Speech Synthesis. In: International Conference on Multimedia Analysis and Pattern Recognition. MAPR, Hanoi, Vietnam, pp. 1\u20135. http:\/\/dx.doi.org\/10.1109\/MAPR53640.2021.9585267.","DOI":"10.1109\/MAPR53640.2021.9585267"},{"key":"10.1016\/j.csl.2026.101985_b92","series-title":"Non-attentive tacotron: Robust and controllable neural TTS synthesis including unsupervised duration modeling","author":"Shen","year":"2021"},{"key":"10.1016\/j.csl.2026.101985_b93","doi-asserted-by":"crossref","unstructured":"Shen, Jonathan, Pang, Ruoming, Weiss, Ron J, Schuster, Mike, Jaitly, Navdeep, Yang, Zongheng, Chen, Zhifeng, Zhang, Yu, Wang, Yuxuan, Skerrv-Ryan, Rj, et al., 2018. Natural TTS synthesis by conditioning wavenet on mel spectrogram predictions. In: IEEE International Conference on Acoustics, Speech and Signal Processing. ICASSP, Calgary, AB, Canada, pp. 4779\u20134783. http:\/\/dx.doi.org\/10.1109\/ICASSP.2018.8461368.","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"10.1016\/j.csl.2026.101985_b94","doi-asserted-by":"crossref","unstructured":"Shin, Yookyung, Lee, Younggun, Jo, Suhee, Hwang, Yeongtae, Kim, Taesu, 2022. Text-driven Emotional Style Control and Cross-speaker Style Transfer in Neural TTS. In: Interspeech. Incheon, Korea, pp. 2313\u20132317. http:\/\/dx.doi.org\/10.21437\/Interspeech.2022-10131.","DOI":"10.21437\/Interspeech.2022-10131"},{"key":"10.1016\/j.csl.2026.101985_b95","unstructured":"\u0160imko, Juraj, T\u00f6r\u00f6, Tuukka, Vainio, Martti, Suni, Antti, 2023. Prosody under control: Controlling prosody in text-to-speech synthesis by adjustments in latent reference space. In: Skarnitzl, Radek, Vol\u00edn, Jan (Eds.), International Congress of Phonetic Sciences. ICPhS, Prague, Czech Republic, pp. 3086\u20133090, URL https:\/\/www.internationalphoneticassociation.org\/icphs-proceedings\/ICPhS2023\/full_papers\/60.pdf."},{"key":"10.1016\/j.csl.2026.101985_b96","unstructured":"Skerry-Ryan, RJ, Battenberg, Eric, Xiao, Ying, Wang, Yuxuan, Stanton, Daisy, Shor, Joel, Weiss, Ron J, Clark, Rob, Saurous, Rif A, 2018. Towards end-to-end prosody transfer for expressive speech synthesis with tacotron. In: International Conference on Machine Learning. ICML, Vol. 80, Stockholmsm\u00e4ssan, Stockholm Sweden, pp. 4693\u20134702, URL https:\/\/proceedings.mlr.press\/v80\/skerry-ryan18a.html."},{"key":"10.1016\/j.csl.2026.101985_b97","doi-asserted-by":"crossref","unstructured":"Sorin, Alexander, Shechtman, Slava, Hoory, Ron, 2020. Principal Style Components: Expressive Style Control and Cross-Speaker Transfer in Neural TTS. In: Interspeech. Shanghai, China, pp. 3411\u20133415. http:\/\/dx.doi.org\/10.21437\/Interspeech.2020-1854.","DOI":"10.21437\/Interspeech.2020-1854"},{"key":"10.1016\/j.csl.2026.101985_b98","doi-asserted-by":"crossref","unstructured":"Stephenson, Brooke, Besacier, Laurent, Girin, Laurent, Hueber, Thomas, 2022. BERT, can HE predict contrastive focus? Predicting and controlling prominence in neural TTS using a language model. In: Interspeech. Incheon, Korea, pp. 3383\u20133387. http:\/\/dx.doi.org\/10.21437\/Interspeech.2022-10116.","DOI":"10.21437\/Interspeech.2022-10116"},{"key":"10.1016\/j.csl.2026.101985_b99","series-title":"LLM theory of mind and alignment: Opportunities and risks","author":"Street","year":"2024"},{"key":"10.1016\/j.csl.2026.101985_b100","doi-asserted-by":"crossref","first-page":"59","DOI":"10.1075\/silv.15.03stu","article-title":"Derhoticisation in Scottish english","volume":"15","author":"Stuart-Smith","year":"2014","journal-title":"Adv. Sociophonetics"},{"key":"10.1016\/j.csl.2026.101985_b101","doi-asserted-by":"crossref","unstructured":"Suni, Antti, Le Maguer, S\u00e9bastien, Kakouros, Sofoklis, T\u00f6r\u00f6, Tuukka, \u0160imko, Juraj, 2025. Style and Prosody control for Zero-shot Speech Synthesis. In: ISCA Speech Synthesis Workshop. Leeuwarden, The Netherlands, pp. 28\u201334. http:\/\/dx.doi.org\/10.21437\/SSW.2025-5.","DOI":"10.21437\/SSW.2025-5"},{"key":"10.1016\/j.csl.2026.101985_b102","doi-asserted-by":"crossref","first-page":"123","DOI":"10.1016\/j.csl.2016.11.001","article-title":"Hierarchical representation and estimation of prosody using continuous wavelet transform","volume":"45","author":"Suni","year":"2017","journal-title":"Comput. Speech Lang."},{"key":"10.1016\/j.csl.2026.101985_b103","doi-asserted-by":"crossref","unstructured":"Taylor, Jason, Richmond, Korin, 2019. Analysis of Pronunciation Learning in End-to-End Speech Synthesis. In: Interspeech. Graz, Austria, pp. 2070\u20132074. http:\/\/dx.doi.org\/10.21437\/Interspeech.2019-2830.","DOI":"10.21437\/Interspeech.2019-2830"},{"issue":"4","key":"10.1016\/j.csl.2026.101985_b104","doi-asserted-by":"crossref","first-page":"84","DOI":"10.3390\/informatics8040084","article-title":"Analysis and assessment of controllability of an expressive deep learning-based tts system","volume":"8","author":"Tits","year":"2021","journal-title":"Informatics"},{"key":"10.1016\/j.csl.2026.101985_b105","first-page":"4475","article-title":"Visualization and interpretation of latent spaces for controlling expressive speech synthesis through audio analysis","author":"Tits","year":"2019","journal-title":"Interspeech"},{"key":"10.1016\/j.csl.2026.101985_b106","series-title":"International Conference on Machine Learning","first-page":"21927","article-title":"Self-supervised models of audio effectively explain human cortical responses to speech","volume":"Vol. 162","author":"Vaidya","year":"2022"},{"key":"10.1016\/j.csl.2026.101985_b107","unstructured":"Valle, Rafael, Shih, Kevin, Prenger, Ryan, Catanzaro, Bryan, 2020. Flowtron: an autoregressive flow-based generative network for text-to-speech synthesis. In: International Conference on Learning Representations. ICLR, Virtual, URL https:\/\/openreview.net\/forum?id=Ig53hpHxS4."},{"key":"10.1016\/j.csl.2026.101985_b108","unstructured":"Van den Oord, Aaron, Dieleman, Sander, Zen, Heiga, Simonyan, Karen, Vinyals, Oriol, Graves, Alex, Kalchbrenner, Nal, Senior, Andrew, Kavukcuoglu, Koray, 2016. WaveNet: A generative model for raw audio. In: ISCA Speech Synthesis Workshop. SSW, Sunnyvale, CA, USA, p. 125, URL https:\/\/www.isca-archive.org\/ssw_2016\/vandenoord16_ssw.html#."},{"key":"10.1016\/j.csl.2026.101985_b109","doi-asserted-by":"crossref","unstructured":"van Rijn, Pol, Mertes, Silvan, Schiller, Dominik, Harrison, Peter M.C., Larrouy-Maestri, Pauline, Andr\u00e9, Elisabeth, Jacoby, Nori, 2021. Exploring Emotional Prototypes in a High Dimensional TTS Latent Space. In: Interspeech. Brno, Czechia, pp. 3870\u20133874. http:\/\/dx.doi.org\/10.21437\/Interspeech.2021-1538.","DOI":"10.21437\/Interspeech.2021-1538"},{"key":"10.1016\/j.csl.2026.101985_b110","unstructured":"Vaswani, Ashish, Shazeer, Noam, Parmar, Niki, Uszkoreit, Jakob, Jones, Llion, Gomez, Aidan N, Kaiser, \u0141ukasz, Polosukhin, Illia, 2017. Attention is all you need. In: Advances in Neural Information Processing Systems. Vol. 30, Long Beach, CA, USA, pp. 5998\u20136008, URL http:\/\/papers.nips.cc\/paper\/7181-attention-is-all-you-need.pdf."},{"key":"10.1016\/j.csl.2026.101985_b111","doi-asserted-by":"crossref","unstructured":"Wagner, Petra, Beskow, Jonas, Betz, Simon, Edlund, Jens, Gustafson, Joakim, Eje Henter, Gustav, Le Maguer, S\u00e9bastien, Malisz, Zofia, Sz\u00e9kely, \u00c9va, T\u00e5nnander, Christina, et al., 2019. Speech synthesis evaluation\u2014state-of-the-art assessment and suggestion for a novel research program. In: ISCA Speech Synthesis Workshop . SSW, http:\/\/dx.doi.org\/10.21437\/SSW.2019-19.","DOI":"10.21437\/SSW.2019-19"},{"key":"10.1016\/j.csl.2026.101985_b112","series-title":"Spark-TTS: An efficient LLM-based text-to-speech model with single-stream decoupled speech tokens","author":"Wang","year":"2025"},{"key":"10.1016\/j.csl.2026.101985_b113","doi-asserted-by":"crossref","unstructured":"Wang, Shuai, Qian, Yanmin, Yu, Kai, 2017. What does the speaker embedding encode?. In: Interspeech. Stockholm, Sweden, pp. 1497\u20131501. http:\/\/dx.doi.org\/10.21437\/Interspeech.2017-1125.","DOI":"10.21437\/Interspeech.2017-1125"},{"key":"10.1016\/j.csl.2026.101985_b114","doi-asserted-by":"crossref","unstructured":"Wang, Yuxuan, Skerry-Ryan, R.J., Stanton, Daisy, Wu, Yonghui, Weiss, Ron J., Jaitly, Navdeep, Yang, Zongheng, Xiao, Ying, Chen, Zhifeng, Bengio, Samy, Le, Quoc, Agiomyrgiannakis, Yannis, Clark, Rob, Saurous, Rif A., 2017. Tacotron: Towards End-to-End Speech Synthesis. In: Interspeech. Stockholm, Sweden, pp. 4006\u20134010. http:\/\/dx.doi.org\/10.21437\/Interspeech.2017-1452.","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"10.1016\/j.csl.2026.101985_b115","unstructured":"Wang, Yuxuan, Stanton, Daisy, Zhang, Yu, Skerry-Ryan, RJ, Battenberg, Eric, Shor, Joel, Xiao, Ying, Ren, Fei, Jia, Ye, Saurous, Rif A, 2018. Style tokens: Unsupervised style modeling, control and transfer in end-to-end speech synthesis. In: International Conference on Machine Learning. ICML, Vol. 80, Stockholmsm\u00e4ssan, Stockholm Sweden, pp. 5180\u20135189, URL http:\/\/proceedings.mlr.press\/v80\/wang18h.html."},{"key":"10.1016\/j.csl.2026.101985_b116","doi-asserted-by":"crossref","unstructured":"Wells, Dan, Tang, Hao, Richmond, Korin, 2022. Phonetic Analysis of Self-supervised Representations of English Speech. In: Interspeech. Incheon, Korea, pp. 3583\u20133587. http:\/\/dx.doi.org\/10.21437\/Interspeech.2022-10884.","DOI":"10.21437\/Interspeech.2022-10884"},{"key":"10.1016\/j.csl.2026.101985_b117","unstructured":"Wichmann, Anne, 2000. The attitudinal effects of prosody, and how they relate to emotion. In: ITRW on Speech and Emotion. Newcastle, Northern Ireland, UK, pp. 143\u2013148, URL https:\/\/www.isca-archive.org\/speechemotion_2000\/wichmann00_speechemotion.html#."},{"key":"10.1016\/j.csl.2026.101985_b118","doi-asserted-by":"crossref","unstructured":"Wu, Pengfei, Ling, Zhenhua, Liu, Lijuan, Jiang, Yuan, Wu, Hongchuan, Dai, Lirong, 2019. End-to-end emotional speech synthesis using style tokens and semi-supervised training. In: Asia-Pacific Signal and Information Processing Association Annual Summit and Conference. APSIPA ASC, Lanzhou, China, pp. 623\u2013627. http:\/\/dx.doi.org\/10.1109\/APSIPAASC47483.2019.9023186.","DOI":"10.1109\/APSIPAASC47483.2019.9023186"},{"key":"10.1016\/j.csl.2026.101985_b119","series-title":"Bigcodec: Pushing the limits of low-bitrate neural speech codec","author":"Xin","year":"2024"},{"issue":"5","key":"10.1016\/j.csl.2026.101985_b120","doi-asserted-by":"crossref","first-page":"1451","DOI":"10.1007\/s11263-020-01429-5","article-title":"Semantic hierarchy emerges in deep generative representations for scene synthesis","volume":"129","author":"Yang","year":"2021","journal-title":"Int. J. Comput. Vis."},{"key":"10.1016\/j.csl.2026.101985_b121","series-title":"ACM\/IEEE International Conference on Human-Robot Interaction","first-page":"1114","article-title":"Butsukusa: A conversational mobile robot describing its own observations and internal states","author":"Yuguchi","year":"2022"},{"key":"10.1016\/j.csl.2026.101985_b122","doi-asserted-by":"crossref","unstructured":"Zeiler, Matthew D., Fergus, Rob, 2014. Visualizing and understanding convolutional networks. In: Computer Vision. ECCV, Zurich, Switzerland, pp. 818\u2013833. http:\/\/dx.doi.org\/10.1007\/978-3-319-10590-1_53.","DOI":"10.1007\/978-3-319-10590-1_53"},{"issue":"11","key":"10.1016\/j.csl.2026.101985_b123","doi-asserted-by":"crossref","first-page":"1039","DOI":"10.1016\/j.specom.2009.04.004","article-title":"Statistical parametric speech synthesis","volume":"51","author":"Zen","year":"2009","journal-title":"Speech Commun."},{"key":"10.1016\/j.csl.2026.101985_b124","doi-asserted-by":"crossref","unstructured":"Zhang, Ya-Jie, Pan, Shifeng, He, Lei, Ling, Zhen-Hua, 2019. Learning latent representations for style control and transfer in end-to-end speech synthesis. In: IEEE International Conference on Acoustics, Speech and Signal Processing . ICASSP, Brighton, UK, pp. 6945\u20136949. http:\/\/dx.doi.org\/10.1109\/ICASSP.2019.8683623.","DOI":"10.1109\/ICASSP.2019.8683623"},{"key":"10.1016\/j.csl.2026.101985_b125","unstructured":"Zhang, Xin, Zhang, Dong, Li, Shimin, Zhou, Yaqian, Qiu, Xipeng, 2024. SpeechTokenizer: Unified Speech Tokenizer for Speech Language Models. In: International Conference on Learning Representations. ICLR, Vienna, Austria, URL https:\/\/openreview.net\/forum?id=AF9Q8Vip84."}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000483?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S0885230826000483?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:13:35Z","timestamp":1779225215000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S0885230826000483"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":125,"alternative-id":["S0885230826000483"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101985","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"A closer look at internal representations of end-to-end Text-to-Speech models: How is phonetic and acoustic information encoded?","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101985","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 The Authors. Published by Elsevier Ltd.","name":"copyright","label":"Copyright"}],"article-number":"101985"}}