{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T22:06:14Z","timestamp":1779228374349,"version":"3.51.4"},"reference-count":48,"publisher":"Elsevier BV","license":[{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/tdm\/userlicense\/1.0\/"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.elsevier.com\/legal\/tdmrep-license"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-017"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-012"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2026,10,1]],"date-time":"2026-10-01T00:00:00Z","timestamp":1790812800000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-004"}],"content-domain":{"domain":["elsevier.com","sciencedirect.com"],"crossmark-restriction":true},"short-container-title":["Computer Speech &amp; Language"],"published-print":{"date-parts":[[2026,10]]},"DOI":"10.1016\/j.csl.2026.101983","type":"journal-article","created":{"date-parts":[[2026,3,19]],"date-time":"2026-03-19T08:37:25Z","timestamp":1773909445000},"page":"101983","update-policy":"https:\/\/doi.org\/10.1016\/elsevier_cm_policy","source":"Crossref","is-referenced-by-count":0,"special_numbering":"C","title":["Assessing the ability of neural TTS systems to model consonant-induced f0 perturbation"],"prefix":"10.1016","volume":"100","author":[{"ORCID":"https:\/\/orcid.org\/0000-0003-0583-0831","authenticated-orcid":false,"given":"Tianle","family":"Yang","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Chengzhe","family":"Sun","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Phil","family":"Rose","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Cassandra L.","family":"Jacobs","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Siwei","family":"Lyu","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"78","reference":[{"key":"10.1016\/j.csl.2026.101983_b1","series-title":"Praat: doing phonetics by computer","author":"Boersma","year":"2007"},{"issue":"4","key":"10.1016\/j.csl.2026.101983_b2","doi-asserted-by":"crossref","first-page":"977","DOI":"10.3758\/BRM.41.4.977","article-title":"Moving beyond Ku\u010dera and Francis: A critical evaluation of current word frequency norms and the introduction of a new and improved word frequency measure for American English","volume":"41","author":"Brysbaert","year":"2009","journal-title":"Behav. Res. Methods"},{"issue":"4","key":"10.1016\/j.csl.2026.101983_b3","doi-asserted-by":"crossref","first-page":"612","DOI":"10.1016\/j.wocn.2011.04.001","article-title":"How does phonology guide phonetics in segment\u2013f0 interaction?","volume":"39","author":"Chen","year":"2011","journal-title":"J. Phon."},{"key":"10.1016\/j.csl.2026.101983_b4","unstructured":"Clements, G.N., Khatiwada, R., 2007. Phonetic realization of contrastively aspirated affricates in Nepali. In: Proceedings of the 16th International Congress of Phonetic Sciences. p. 632."},{"key":"10.1016\/j.csl.2026.101983_b5","series-title":"The Corpus of Contemporary American English (COCA), Version 2.2","author":"Davies","year":"2015"},{"issue":"3","key":"10.1016\/j.csl.2026.101983_b6","doi-asserted-by":"crossref","first-page":"474","DOI":"10.1353\/lan.0.0035","article-title":"Time and thyme are not homophones: The effect of lemma frequency on word durations in spontaneous speech","volume":"84","author":"Gahl","year":"2008","journal-title":"Language"},{"issue":"1","key":"10.1016\/j.csl.2026.101983_b7","doi-asserted-by":"crossref","first-page":"124","DOI":"10.1353\/lan.2024.a922001","article-title":"Laryngeal contrast and sound change: The production and perception of plosive voicing and co-intrinsic pitch","volume":"100","author":"Gao","year":"2024","journal-title":"Language"},{"key":"10.1016\/j.csl.2026.101983_b8","first-page":"198","article-title":"A note on laryngeal features","volume":"101","author":"Halle","year":"1971","journal-title":"Q. Prog. Rep."},{"issue":"1","key":"10.1016\/j.csl.2026.101983_b9","doi-asserted-by":"crossref","first-page":"425","DOI":"10.1121\/1.3021306","article-title":"Effects of obstruent consonants on fundamental frequency at vowel onset in English","volume":"125","author":"Hanson","year":"2009","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101983_b10","series-title":"Where do Phonological Features Come from? Cognitive, Physical and Developmental Bases of Distinctive Speech Categories","first-page":"131","article-title":"Automaticity vs. feature-enhancement in the control of segmental F0","author":"Hoole","year":"2011"},{"key":"10.1016\/j.csl.2026.101983_b11","doi-asserted-by":"crossref","unstructured":"Huang, R., Zhang, C., Ren, Y., Zhao, Z., Yu, D., 2023. Prosody-tts: Improving prosody with masked autoencoder and conditional diffusion model for expressive text-to-speech. In: Findings of the Association for Computational Linguistics: ACL 2023. pp. 8018\u20138034.","DOI":"10.18653\/v1\/2023.findings-acl.508"},{"key":"10.1016\/j.csl.2026.101983_b12","series-title":"The LJ speech dataset","author":"Ito","year":"2017"},{"issue":"4","key":"10.1016\/j.csl.2026.101983_b13","doi-asserted-by":"crossref","first-page":"EL405","DOI":"10.1121\/1.4934178","article-title":"Intrinsic fundamental frequency of vowels is moderated by regional dialect","volume":"138","author":"Jacewicz","year":"2015","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101983_b14","series-title":"International Conference on Machine Learning","first-page":"5530","article-title":"Conditional variational autoencoder with adversarial learning for end-to-end text-to-speech","author":"Kim","year":"2021"},{"issue":"3","key":"10.1016\/j.csl.2026.101983_b15","doi-asserted-by":"crossref","first-page":"419","DOI":"10.2307\/416481","article-title":"Phonetic knowledge","volume":"70","author":"Kingston","year":"1994","journal-title":"Language"},{"issue":"4","key":"10.1016\/j.csl.2026.101983_b16","doi-asserted-by":"crossref","first-page":"2400","DOI":"10.1121\/1.4962445","article-title":"Effects of obstruent voicing on vowel F0: Evidence from \u201ctrue voicing\u201d languages","volume":"140","author":"Kirby","year":"2016","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101983_b17","first-page":"17022","article-title":"Hifi-gan: Generative adversarial networks for efficient and high fidelity speech synthesis","volume":"33","author":"Kong","year":"2020","journal-title":"Adv. Neural Inf. Process. Syst."},{"issue":"6","key":"10.1016\/j.csl.2026.101983_b18","doi-asserted-by":"crossref","first-page":"449","DOI":"10.1016\/S0021-9924(03)00032-7","article-title":"Perceptual effects of a flattened fundamental frequency at the sentence level under different listening conditions","volume":"36","author":"Laures","year":"2003","journal-title":"J. Commun. Disord."},{"issue":"3","key":"10.1016\/j.csl.2026.101983_b19","doi-asserted-by":"crossref","first-page":"1314","DOI":"10.1121\/1.397462","article-title":"The cricothyroid muscle in voicing control","volume":"85","author":"L\u00f6fqvist","year":"1989","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101983_b20","doi-asserted-by":"crossref","unstructured":"McAuliffe, M., Socolof, M., Mihuc, S., Wagner, M., Sonderegger, M., 2017a. Montreal forced aligner: Trainable text-speech alignment using kaldi. In: Interspeech. pp. 498\u2013502.","DOI":"10.21437\/Interspeech.2017-1386"},{"key":"10.1016\/j.csl.2026.101983_b21","series-title":"English (US) MFA Dictionary v2.0.0a","author":"McAuliffe","year":"2022"},{"key":"10.1016\/j.csl.2026.101983_b22","series-title":"Interspeech 2017","first-page":"3887","article-title":"Polyglot and speech corpus tools: A system for representing, integrating, and querying speech corpora","author":"McAuliffe","year":"2017"},{"key":"10.1016\/j.csl.2026.101983_b23","series-title":"Quantifying privacy risks of masked language models using membership inference attacks","author":"Mireshghallah","year":"2022"},{"key":"10.1016\/j.csl.2026.101983_b24","doi-asserted-by":"crossref","unstructured":"M\u00fcller, N.M., Czempin, P., Dieckmann, F., Froghyar, A., B\u00f6ttinger, K., 2022. Does audio deepfake detection generalize?. In: Interspeech.","DOI":"10.21437\/Interspeech.2022-108"},{"issue":"5","key":"10.1016\/j.csl.2026.101983_b25","doi-asserted-by":"crossref","first-page":"603","DOI":"10.1016\/j.jvoice.2018.04.004","article-title":"The effects of stress type, vowel identity, baseline f0, and loudness on the relative fundamental frequency of individuals with healthy voices","volume":"33","author":"Park","year":"2019","journal-title":"J. Voice"},{"key":"10.1016\/j.csl.2026.101983_b26","series-title":"Fastspeech 2: Fast and high-quality end-to-end text to speech","author":"Ren","year":"2020"},{"key":"10.1016\/j.csl.2026.101983_b27","unstructured":"Rose, P., Yang, T., 2022. Modelling Interaction between tone and phonation type in the Northern Wu dialect of Jinshan. In: Proc. 18th Int\u2019l Australasian Conf. on Speech Science & Technology. pp. 221\u2013225."},{"key":"10.1016\/j.csl.2026.101983_b28","doi-asserted-by":"crossref","DOI":"10.1016\/j.wocn.2020.100979","article-title":"Acoustic cues in production and perception of the four-way stop laryngeal contrast in Hindi and Urdu","volume":"81","author":"Schertz","year":"2020","journal-title":"J. Phon."},{"key":"10.1016\/j.csl.2026.101983_b29","series-title":"2018 IEEE International Conference on Acoustics, Speech and Signal Processing","first-page":"4779","article-title":"Natural tts synthesis by conditioning wavenet on mel spectrogram predictions","author":"Shen","year":"2018"},{"key":"10.1016\/j.csl.2026.101983_b30","doi-asserted-by":"crossref","first-page":"217","DOI":"10.1016\/j.wocn.2017.10.002","article-title":"Articulatory adjustments in initial voiced stops in Spanish, French and English","volume":"66","author":"Sol\u00e9","year":"2018","journal-title":"J. Phon."},{"key":"10.1016\/j.csl.2026.101983_b31","doi-asserted-by":"crossref","unstructured":"Song, C., Shmatikov, V., 2019. Auditing data provenance in text-generation models. In: Proceedings of the 25th ACM SIGKDD International Conference on Knowledge Discovery & Data Mining. pp. 196\u2013206.","DOI":"10.1145\/3292500.3330885"},{"key":"10.1016\/j.csl.2026.101983_b32","series-title":"Generalised additive mixed models for dynamic analysis in linguistics: A practical introduction","author":"S\u00f3skuthy","year":"2017"},{"issue":"1","key":"10.1016\/j.csl.2026.101983_b33","doi-asserted-by":"crossref","first-page":"1","DOI":"10.1353\/lan.2025.a954227","article-title":"The crosslinguistic distribution of vowel and consonant intrinsic F0 effects","volume":"101","author":"Ting","year":"2025","journal-title":"Language"},{"key":"10.1016\/j.csl.2026.101983_b34","series-title":"Wavenet: A generative model for raw audio","first-page":"1","author":"Van Den Oord","year":"2016"},{"issue":"2","key":"10.1016\/j.csl.2026.101983_b35","doi-asserted-by":"crossref","first-page":"168","DOI":"10.1016\/j.wocn.2011.02.007","article-title":"Intrinsic vowel F0, the size of vowel inventories and second language acquisition","volume":"39","author":"Van Hoof","year":"2011","journal-title":"J. Phon."},{"key":"10.1016\/j.csl.2026.101983_b36","series-title":"Itsadug: Interpreting time series and autocorrelated data using GAMMs","author":"Van Rij","year":"2015"},{"key":"10.1016\/j.csl.2026.101983_b37","series-title":"Tacotron: Towards end-to-end speech synthesis","author":"Wang","year":"2017"},{"issue":"3","key":"10.1016\/j.csl.2026.101983_b38","doi-asserted-by":"crossref","first-page":"349","DOI":"10.1016\/S0095-4470(95)80165-0","article-title":"The universality of intrinsic F0 of vowels","volume":"23","author":"Whalen","year":"1995","journal-title":"J. Phon."},{"issue":"4","key":"10.1016\/j.csl.2026.101983_b39","doi-asserted-by":"crossref","first-page":"2533","DOI":"10.1121\/1.411973","article-title":"Intrinsic F0 of vowels in the babbling of 6-, 9-, and 12-month-old French-and English-learning infants","volume":"97","author":"Whalen","year":"1995","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101983_b40","doi-asserted-by":"crossref","first-page":"86","DOI":"10.1016\/j.wocn.2018.03.002","article-title":"Analyzing dynamic phonetic data using generalized additive mixed modeling: A tutorial focusing on articulatory differences between L1 and L2 speakers of English","volume":"70","author":"Wieling","year":"2018","journal-title":"J. Phon."},{"issue":"5","key":"10.1016\/j.csl.2026.101983_b41","doi-asserted-by":"crossref","first-page":"2751","DOI":"10.1121\/1.4896471","article-title":"A cross-dialectal acoustic comparison of vowels in Northern and Southern British English","volume":"136","author":"Williams","year":"2014","journal-title":"J. Acoust. Soc. Am."},{"key":"10.1016\/j.csl.2026.101983_b42","series-title":"Generalized Additive Models: An Introduction with R","author":"Wood","year":"2017"},{"issue":"2","key":"10.1016\/j.csl.2026.101983_b43","doi-asserted-by":"crossref","first-page":"165","DOI":"10.1017\/S0025100303001270","article-title":"Effects of consonant aspiration on Mandarin tones","volume":"33","author":"Xu","year":"2003","journal-title":"J. Int. Phon. Assoc."},{"issue":"4","key":"10.1016\/j.csl.2026.101983_b44","doi-asserted-by":"crossref","first-page":"2877","DOI":"10.1121\/10.0004239","article-title":"Consonantal F0 perturbation in American English involves multiple mechanisms","volume":"149","author":"Xu","year":"2021","journal-title":"J. Acoust. Soc. Am."},{"issue":"1","key":"10.1016\/j.csl.2026.101983_b45","doi-asserted-by":"crossref","first-page":"5895","DOI":"10.3765\/plsa.v10i1.5895","article-title":"Onset-tone interaction in Mundabli","volume":"10","author":"Yang","year":"2025","journal-title":"Proc. Linguist. Soc. Am."},{"key":"10.1016\/j.csl.2026.101983_b46","article-title":"Forensic deepfake audio detection using segmental speech features","volume":"379","author":"Yang","year":"2025","journal-title":"Forensic Sci. Int."},{"key":"10.1016\/j.csl.2026.101983_b47","series-title":"Clapspeech: Learning prosody from text context with contrastive language-audio pre-training","author":"Ye","year":"2023"},{"issue":"4","key":"10.1016\/j.csl.2026.101983_b48","doi-asserted-by":"crossref","first-page":"2472","DOI":"10.1121\/1.5146842","article-title":"Effect of consonants on onset F0: Evidence from Kansai Japanese","volume":"148","author":"Zhang","year":"2020","journal-title":"J. Acoust. Soc. Am."}],"container-title":["Computer Speech &amp; Language"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S088523082600046X?httpAccept=text\/xml","content-type":"text\/xml","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/api.elsevier.com\/content\/article\/PII:S088523082600046X?httpAccept=text\/plain","content-type":"text\/plain","content-version":"vor","intended-application":"text-mining"}],"deposited":{"date-parts":[[2026,5,19]],"date-time":"2026-05-19T21:12:17Z","timestamp":1779225137000},"score":1,"resource":{"primary":{"URL":"https:\/\/linkinghub.elsevier.com\/retrieve\/pii\/S088523082600046X"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,10]]},"references-count":48,"alternative-id":["S088523082600046X"],"URL":"https:\/\/doi.org\/10.1016\/j.csl.2026.101983","relation":{},"ISSN":["0885-2308"],"issn-type":[{"value":"0885-2308","type":"print"}],"subject":[],"published":{"date-parts":[[2026,10]]},"assertion":[{"value":"Elsevier","name":"publisher","label":"This article is maintained by"},{"value":"Assessing the ability of neural TTS systems to model consonant-induced f0 perturbation","name":"articletitle","label":"Article Title"},{"value":"Computer Speech & Language","name":"journaltitle","label":"Journal Title"},{"value":"https:\/\/doi.org\/10.1016\/j.csl.2026.101983","name":"articlelink","label":"CrossRef DOI link to publisher maintained version"},{"value":"article","name":"content_type","label":"Content Type"},{"value":"\u00a9 2026 Elsevier Ltd. All rights are reserved, including those for text and data mining, AI training, and similar technologies.","name":"copyright","label":"Copyright"}],"article-number":"101983"}}