{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T03:57:46Z","timestamp":1784779066495,"version":"3.55.0"},"reference-count":47,"publisher":"Springer Science and Business Media LLC","issue":"6","license":[{"start":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T00:00:00Z","timestamp":1634515200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2021,10,18]],"date-time":"2021-10-18T00:00:00Z","timestamp":1634515200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["J Ambient Intell Human Comput"],"published-print":{"date-parts":[[2023,6]]},"DOI":"10.1007\/s12652-021-03542-w","type":"journal-article","created":{"date-parts":[[2021,10,19]],"date-time":"2021-10-19T00:07:34Z","timestamp":1634602054000},"page":"6751-6768","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":18,"title":["Generating synthetic dysarthric speech to overcome dysarthria acoustic data scarcity"],"prefix":"10.1007","volume":"14","author":[{"given":"Andrew","family":"Hu","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Dhruv","family":"Phadnis","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0003-1543-5931","authenticated-orcid":false,"given":"Seyed Reza","family":"Shahamiri","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2021,10,18]]},"reference":[{"key":"3542_CR1","unstructured":"Arik SO, Chrzanowski M, Coates A, Diamos G, Gibiansky A, Kang Y et al (2017) Deep voice: real-time neural text-to-speech. ArXiv, Retrieved from https:\/\/arxiv.org\/abs\/1702.07825v2"},{"key":"3542_CR2","doi-asserted-by":"crossref","unstructured":"Bennett CL (2005) Large scale evaluation of corpus-based synthesizers: results and lessons from the blizzard challenge 2005. In: 9th European conference on speech communication and technology, pp 105\u2013108","DOI":"10.21437\/Interspeech.2005-79"},{"key":"3542_CR4","doi-asserted-by":"crossref","unstructured":"Black AW (2003) Unit selection and emotional speech. In: EUROSPEECH 2003\u20148th European conference on speech communication and technology, vol 3, pp 1649\u20131652","DOI":"10.21437\/Eurospeech.2003-473"},{"key":"3542_CR3","doi-asserted-by":"crossref","unstructured":"Black A, Campbell N (1996) Optimising selection of units from speech databases for concatenative synthesis. International Speech Communication Association, 1","DOI":"10.21437\/Eurospeech.1995-148"},{"key":"3542_CR5","doi-asserted-by":"crossref","unstructured":"Christensen H, Cunningham SP, Fox C, Green PD, Hain T (2012) A comparative study of adaptive, automatic recognition of disordered speech. Paper presented at the INTERSPEECH 2012, pp 1776\u20131779","DOI":"10.21437\/Interspeech.2012-484"},{"key":"3542_CR500","doi-asserted-by":"publisher","unstructured":"Demonte P (2019) HARVARD corpus speech shaped noise and speech modulated noise for SIN test. Collection. https:\/\/doi.org\/10.17866\/rd.salford.c.4700054.v1","DOI":"10.17866\/rd.salford.c.4700054.v1"},{"key":"3542_CR6","unstructured":"Donahue C, McAuley J, Puckette M (2019) Adversarial audio synthesis. In: 7th International conference on learning representations, ICLR 2019, pp 1\u201316"},{"key":"3542_CR9","doi-asserted-by":"publisher","unstructured":"Dua M, Aggarwal RK, Biswas M (2017) Discriminative training using heterogeneous feature vector for hindi automatic speech recognition system. In: Paper presented at the 2017 international conference on computer and applications (ICCA), pp 158\u2013162. https:\/\/doi.org\/10.1109\/COMAPP.2017.8079777","DOI":"10.1109\/COMAPP.2017.8079777"},{"issue":"3","key":"3542_CR7","doi-asserted-by":"publisher","first-page":"389","DOI":"10.1016\/j.jestch.2018.04.005","volume":"21","author":"M Dua","year":"2018","unstructured":"Dua M, Aggarwal RK, Biswas M (2018) Performance evaluation of hindi speech recognition system using optimized filterbanks. Eng Sci Technol Int J 21(3):389\u2013398. https:\/\/doi.org\/10.1016\/j.jestch.2018.04.005","journal-title":"Eng Sci Technol Int J"},{"issue":"6","key":"3542_CR8","doi-asserted-by":"publisher","first-page":"2301","DOI":"10.1007\/s12652-018-0828-x","volume":"10","author":"M Dua","year":"2019","unstructured":"Dua M, Aggarwal RK, Biswas M (2019) GFCC based discriminatively trained noise robust continuous ASR system for hindi language. J Ambient Intell Humaniz Comput 10(6):2301\u20132314. https:\/\/doi.org\/10.1007\/s12652-018-0828-x","journal-title":"J Ambient Intell Humaniz Comput"},{"key":"3542_CR10","unstructured":"Duffy JR (2005) Motor speech disorders: substrates, differential diagnosis, and management Mosby"},{"key":"3542_CR11","doi-asserted-by":"publisher","unstructured":"Ellis D, Morgan N (1999) Size matters: an empirical study of neural network training for large vocabulary continuous speech recognition. In: ICASSP, IEEE international conference on acoustics, speech and signal processing\u2014proceedings, vol 2, pp 1013\u20131016. https:\/\/doi.org\/10.1109\/icassp.1999.759875","DOI":"10.1109\/icassp.1999.759875"},{"issue":"3","key":"3542_CR12","doi-asserted-by":"publisher","first-page":"165","DOI":"10.3109\/13682828009112541","volume":"15","author":"P Enderby","year":"1980","unstructured":"Enderby P (1980) Frenchay dysarthria assessment. Int J Lang Commun Disord 15(3):165\u2013173. https:\/\/doi.org\/10.3109\/13682828009112541","journal-title":"Int J Lang Commun Disord"},{"key":"3542_CR13","unstructured":"Fang W, Chung Y-A, Glass J (2019) Towards transfer learning for end-to-end speech synthesis from deep pre-trained language models. arXiv:1906.07307"},{"issue":"4","key":"3542_CR14","doi-asserted-by":"publisher","first-page":"571","DOI":"10.1016\/j.compbiomed.2006.08.008","volume":"37","author":"ES Fonseca","year":"2007","unstructured":"Fonseca ES, Guido RC, Scalassara PR, Maciel CD, Pereira JC (2007) Wavelet time-frequency analysis and least squares support vector machines for the identification of voice disorders. Comput Biol Med 37(4):571\u2013578. https:\/\/doi.org\/10.1016\/j.compbiomed.2006.08.008","journal-title":"Comput Biol Med"},{"key":"3542_CR15","doi-asserted-by":"publisher","first-page":"105","DOI":"10.1016\/j.neunet.2021.02.008","volume":"139","author":"S Gupta","year":"2021","unstructured":"Gupta S, Patil AT, Purohit M, Parmar M, Patel M, Patil HA, Guido RC (2021) Residual neural network precisely quantifies dysarthria severity-level based on short-duration speech segments. Neural Netw 139:105\u2013117","journal-title":"Neural Netw"},{"key":"3542_CR16","doi-asserted-by":"crossref","unstructured":"Hsu CC, Hwang HT, Wu YC, Tsao Y, Wang HM (2017) Voice conversion from unaligned corpora using variational autoencoding wasserstein generative adversarial networks. In: Paper presented at the Nterspeech 2017, 2017-Augus, pp 3364\u20133368","DOI":"10.21437\/Interspeech.2017-63"},{"key":"3542_CR17","unstructured":"Hunt AJ, Black AW (1996) Unit selection in a concatenative speech synthesis system using a large speech database. In: ICASSP, IEEE international conference on acoustics, speech and signal processing\u2014proceedings, vol 1, pp 373-376"},{"key":"3542_CR18","unstructured":"Ito K, Johnson L (2017) The LJ speech dataset. Retrieved from https:\/\/keithito.com\/LJ-Speech-Dataset\/"},{"key":"3542_CR19","doi-asserted-by":"publisher","unstructured":"Jiao Y, Tu M, Berisha V, Liss J (2018) Simulating dysarthric speech for training data augmentation in clinical speech applications. In: 2018 IEEE international conference on acoustics, speech and signal processing (ICASSP), pp 6009\u20136013. https:\/\/doi.org\/10.1109\/ICASSP.2018.8462290","DOI":"10.1109\/ICASSP.2018.8462290"},{"key":"3542_CR20","doi-asserted-by":"crossref","unstructured":"Kim H, Hasegawa-Johnson M, Perlman A, Gunderson J, Huang T, Watkin K, Frame S (2008) Dysarthric speech database for universal access research. In: Paper presented at the INTERSPEECH 2008\u20149th annual conference of the international speech communication association, pp 1741\u20131744","DOI":"10.21437\/Interspeech.2008-480"},{"key":"3542_CR25","unstructured":"Kyubyong Park, & Mulc T (2018) A TensorFlow implementation of tacotron: a fully end-to-end text-to-speech synthesis model. Retrieved from https:\/\/github.com\/Kyubyong\/tacotron"},{"key":"3542_CR21","doi-asserted-by":"crossref","unstructured":"Menendez-Pidal X, Polikoff JB, Peters SM, Leonzio JE, Bunnell HT (1996) Nemours database of dysarthric speech. In: International conference on spoken language processing, ICSLP, proceedings, vol 3, pp 1962\u20131965","DOI":"10.21437\/ICSLP.1996-503"},{"key":"3542_CR22","doi-asserted-by":"publisher","first-page":"144","DOI":"10.1016\/j.neucom.2011.09.037","volume":"100","author":"M Mubashir","year":"2013","unstructured":"Mubashir M, Shao L, Seed L (2013) Audio augmentation for speech recognition tom. Neurocomputing 100:144\u2013152. https:\/\/doi.org\/10.1016\/j.neucom.2011.09.037","journal-title":"Neurocomputing"},{"key":"3542_CR23","unstructured":"Oord AVD, Dieleman S, Zen H, Simonyan K, Vinyals O, Graves A et al (2016) WaveNet: a generative model for raw audio, pp 1\u201315. arXiv:1609.03499"},{"key":"3542_CR24","doi-asserted-by":"crossref","unstructured":"Panayotov V, Chen G, Povey D, Khudanpur S (2015) Librispeech: an ASR corpus based on public domain audio books. In: IEEE international conference on acoustics, speech and signal processing\u2014proceedings, 2015-Augus, pp 5206\u20135210","DOI":"10.1109\/ICASSP.2015.7178964"},{"key":"3542_CR26","unstructured":"Ping W, Peng K, Gibiansky A, Arik S, Kannan A, Narang S et al. (2017) Deep voice 3: 2000-speaker neural text-to-speech [computer software]"},{"issue":"4","key":"3542_CR27","doi-asserted-by":"publisher","first-page":"523","DOI":"10.1007\/s10579-011-9145-0","volume":"46","author":"F Rudzicz","year":"2012","unstructured":"Rudzicz F, Namasivayam AK, Wolff T (2012) The TORGO database of acoustic and articulatory speech from speakers with dysarthria. Lang Resour Eval 46(4):523\u2013541","journal-title":"Lang Resour Eval"},{"issue":"3","key":"3542_CR28","doi-asserted-by":"publisher","first-page":"279","DOI":"10.1109\/LSP.2017.2657381","volume":"24","author":"J Salamon","year":"2017","unstructured":"Salamon J, Bello JP (2017) Deep convolutional neural networks and data augmentation for environmental sound classification. IEEE Signal Process Lett 24(3):279\u2013283","journal-title":"IEEE Signal Process Lett"},{"issue":"6","key":"3542_CR29","doi-asserted-by":"publisher","first-page":"18","DOI":"10.1109\/MSP.2012.2197156","volume":"29","author":"G Saon","year":"2012","unstructured":"Saon G, Chien JT (2012) Large-vocabulary continuous speech recognition systems: a look at some recent advances. IEEE Signal Process Mag 29(6):18\u201333. https:\/\/doi.org\/10.1109\/MSP.2012.2197156","journal-title":"IEEE Signal Process Mag"},{"key":"3542_CR30","doi-asserted-by":"publisher","unstructured":"Sehgal S, Cunningham S (2015) Model adaptation and adaptive training for the recognition of dysarthric speech. In: Paper presented at the proceedings of SLPAT 2015: 6th workshop on speech and language processing for assistive technologies, pp 65\u201371. https:\/\/doi.org\/10.18653\/v1\/W15-5112 Retrieved from http:\/\/aclweb.org\/anthology\/W15-5112","DOI":"10.18653\/v1\/W15-5112"},{"key":"3542_CR31","doi-asserted-by":"publisher","first-page":"852","DOI":"10.1109\/TNSRE.2021.3076778","volume":"29","author":"SR Shahamiri","year":"2021","unstructured":"Shahamiri SR (2021) Speech vision: an end-to-end deep learning-based dysarthric automatic speech recognition system. IEEE Trans Neural Syst Rehabil Eng 29:852\u2013861. https:\/\/doi.org\/10.1109\/TNSRE.2021.3076778","journal-title":"IEEE Trans Neural Syst Rehabil Eng"},{"key":"3542_CR32","doi-asserted-by":"publisher","DOI":"10.1016\/j.aei.2014.01.001","author":"SR Shahamiri","year":"2014","unstructured":"Shahamiri SR, Binti Salim SS (2014a) Artificial neural networks as speech recognisers for dysarthric speech: identifying the best-performing set of MFCC parameters and studying a speaker-independent approach. Adv Eng Inf. https:\/\/doi.org\/10.1016\/j.aei.2014.01.001","journal-title":"Adv Eng Inf"},{"issue":"5","key":"3542_CR33","doi-asserted-by":"publisher","first-page":"1053","DOI":"10.1109\/TNSRE.2014.2309336","volume":"22","author":"SR Shahamiri","year":"2014","unstructured":"Shahamiri SR, Binti Salim SS (2014b) A multi-views multi-learners approach towards dysarthric speech recognition using multi-nets artificial neural networks. IEEE Trans Neural Syst Rehabil Eng 22(5):1053\u20131063. https:\/\/doi.org\/10.1109\/TNSRE.2014.2309336","journal-title":"IEEE Trans Neural Syst Rehabil Eng"},{"key":"3542_CR34","doi-asserted-by":"crossref","unstructured":"Shen J, Pang R, Weiss RJ, Schuster M, Jaitly N, Yang Z et al (2018) Natural TTS synthesis by conditioning wavenet on MEL spectrogram predictions. In: ICASSP, IEEE international conference on acoustics, speech and signal processing\u2014proceedings, 2018-April, pp 4779\u20134783","DOI":"10.1109\/ICASSP.2018.8461368"},{"key":"3542_CR35","unstructured":"Sotelo J, Mehri S, Kumar K, Santos JF, Kastner K, Courville A, Bengio Y (2019) Char2Wav: end-to-end speech synthesis. In: 5th international conference on learning representations, ICLR 2017\u2014workshop track proceedings, (2015), pp 1\u20136"},{"key":"3542_CR36","doi-asserted-by":"publisher","unstructured":"Tachibana H, Uenoyama K, Aihara S (2018) Efficiently trainable text-to-speech system based on deep convolutional networks with guided attention. In: Paper presented at the ICASSP, IEEE international conference on acoustics, speech and signal processing\u2014proceedings 2018-April pp 4784\u20134788. https:\/\/doi.org\/10.1109\/ICASSP.2018.8461829","DOI":"10.1109\/ICASSP.2018.8461829"},{"key":"3542_CR37","doi-asserted-by":"publisher","DOI":"10.1017\/CBO9780511816338","volume-title":"Text-to-speech synthesis","author":"P Taylor","year":"2009","unstructured":"Taylor P (2009) Text-to-speech synthesis. Cambridge University Press, New York"},{"key":"3542_CR38","doi-asserted-by":"publisher","unstructured":"Tirumala SS, Shahamiri SR (2016) A review on deep learning approaches in speaker identification. In: Paper presented at the proceedings of the 8th international conference on signal processing systems\u2014ICSPS 2016, pp 142\u2013147. https:\/\/doi.org\/10.1145\/3015166.3015210. Retrieved from http:\/\/dl.acm.org\/citation.cfm?doid=3015166.3015210","DOI":"10.1145\/3015166.3015210"},{"key":"3542_CR39","doi-asserted-by":"publisher","first-page":"EL416","DOI":"10.1121\/1.4967208","volume":"140","author":"M Tu","year":"2016","unstructured":"Tu M, Wisler A, Berisha V, Liss JM (2016) The relationship between perceptual disturbances in dysarthric speech and automatic speech recognition performance. J Acoust Soc Am 140:EL416\u2013EL422","journal-title":"J Acoust Soc Am"},{"key":"3542_CR40","doi-asserted-by":"publisher","unstructured":"Vachhani B, Bhat C, Kopparapu SK (2018) Data augmentation using healthy speech for dysarthric speech recognition. In: Paper presented at the Interspeech 2018, pp 471\u2013475. https:\/\/doi.org\/10.21437\/Interspeech.2018-1751. Retrieved from http:\/\/www.isca-speech.org\/archive\/Interspeech_2018\/abstracts\/1751.html","DOI":"10.21437\/Interspeech.2018-1751"},{"key":"3542_CR41","doi-asserted-by":"publisher","unstructured":"Wang Y, Skerry-Ryan RJ, Stanton D, Wu Y, Weiss RJ, Jaitly N, Le Q (2017) Tacotron: towards end-to-end speech synthesis. In: Proceedings of the annual conference of the international speech communication association, INTERSPEECH, 2017-Augus, pp 4006\u20134010. https:\/\/doi.org\/10.21437\/Interspeech.2017-1452","DOI":"10.21437\/Interspeech.2017-1452"},{"key":"3542_CR42","unstructured":"Yamamoto R (2018) Deepvoice3_pytorch [computer software]"},{"key":"3542_CR43","doi-asserted-by":"publisher","unstructured":"Ypma A, Heskes T (2017) Multi-stage DNN training for automatic recognition of dysarthric speech. In: Paper presented at the Interspeech 2017, https:\/\/doi.org\/10.21437\/Interspeech.2017-303","DOI":"10.21437\/Interspeech.2017-303"},{"issue":"5","key":"3542_CR44","doi-asserted-by":"publisher","first-page":"1071","DOI":"10.1109\/TASL.2010.2076805","volume":"19","author":"K Yu","year":"2011","unstructured":"Yu K, Young S (2011) Continuous F0 modeling for HMM based statistical parametric speech synthesis. IEEE Trans Audio Speech Lang Process 19(5):1071\u20131079. https:\/\/doi.org\/10.1109\/TASL.2010.2076805","journal-title":"IEEE Trans Audio Speech Lang Process"},{"issue":"11","key":"3542_CR45","first-page":"1229","volume":"51","author":"H Zen","year":"2007","unstructured":"Zen H, Tokuda K, Black AW (2007) Statistical parametric speech synthesis. Speech Commun 51(11):1229\u20131232","journal-title":"Speech Commun"},{"key":"3542_CR46","doi-asserted-by":"publisher","first-page":"1645","DOI":"10.1109\/TASL.2007.899236","volume":"15","author":"X Zhu","year":"2007","unstructured":"Zhu X, Beauregard G, Wyse L (2007) Real-time signal estimation from modified short-time fourier transform magnitude spectra. IEEE Trans Audio Speech Lang Process 15:1645\u20131653","journal-title":"IEEE Trans Audio Speech Lang Process"}],"container-title":["Journal of Ambient Intelligence and Humanized Computing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s12652-021-03542-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s12652-021-03542-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s12652-021-03542-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2023,5,23]],"date-time":"2023-05-23T17:55:42Z","timestamp":1684864542000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s12652-021-03542-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,10,18]]},"references-count":47,"journal-issue":{"issue":"6","published-print":{"date-parts":[[2023,6]]}},"alternative-id":["3542"],"URL":"https:\/\/doi.org\/10.1007\/s12652-021-03542-w","relation":{},"ISSN":["1868-5137","1868-5145"],"issn-type":[{"value":"1868-5137","type":"print"},{"value":"1868-5145","type":"electronic"}],"subject":[],"published":{"date-parts":[[2021,10,18]]},"assertion":[{"value":"5 January 2021","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 October 2021","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"18 October 2021","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}