{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T09:47:55Z","timestamp":1785577675449,"version":"3.56.0"},"reference-count":74,"publisher":"Springer Science and Business Media LLC","issue":"2","license":[{"start":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T00:00:00Z","timestamp":1717200000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,6,1]],"date-time":"2024-06-01T00:00:00Z","timestamp":1717200000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,6]]},"DOI":"10.1007\/s10772-024-10120-w","type":"journal-article","created":{"date-parts":[[2024,7,5]],"date-time":"2024-07-05T14:23:17Z","timestamp":1720189397000},"page":"483-502","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":9,"title":["Pathological voice classification system based on CNN-BiLSTM network using speech enhancement and multi-stream approach"],"prefix":"10.1007","volume":"27","author":[{"ORCID":"https:\/\/orcid.org\/0009-0001-7119-2981","authenticated-orcid":false,"given":"Soumeya","family":"Belabbas","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Djamel","family":"Addou","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sid Ahmed","family":"Selouani","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,7,5]]},"reference":[{"key":"10120_CR1","doi-asserted-by":"publisher","unstructured":"Albawi, S., Mohammed, T. A., & Al-Zawi, S. (2017). Understanding of a convolutional neural network. In International conference on engineering and technology (ICET) (pp. 1\u20136). https:\/\/doi.org\/10.1109\/ICEngTechnol.2017.8308186.","DOI":"10.1109\/ICEngTechnol.2017.8308186"},{"key":"10120_CR2","doi-asserted-by":"publisher","first-page":"46474","DOI":"10.1109\/ACCESS.2019.2905597","volume":"7","author":"M Alhussein","year":"2019","unstructured":"Alhussein, M., & Muhammad, G. (2019). Automatic voice pathology monitoring using parallel deep models for smart healthcare. IEEE Access, 7, 46474\u201346479. https:\/\/doi.org\/10.1109\/ACCESS.2019.2905597","journal-title":"IEEE Access"},{"issue":"3","key":"10120_CR3","doi-asserted-by":"publisher","first-page":"1061","DOI":"10.18576\/amis\/100324","volume":"10","author":"F Amara","year":"2016","unstructured":"Amara, F., Fezari, M., & Bourouba, H. (2016). An improved GMM-SVM system based on distance metric for voice pathology detection. An International Journal of Applied Mathematics & Information Sciences, 10(3), 1061\u20131070. https:\/\/doi.org\/10.18576\/amis\/100324","journal-title":"An International Journal of Applied Mathematics & Information Sciences"},{"key":"10120_CR4","unstructured":"American Speech-Language-Hearing Association. (1993). Definitions of communication disorders and variations [relevant paper]. Retrieved from https:\/\/www.asha.org\/policy\/rp1993-00208\/."},{"key":"10120_CR5","doi-asserted-by":"publisher","first-page":"107310","DOI":"10.1016\/j.asoc.2021.107310","volume":"106","author":"H Ank\u0131\u015fhan","year":"2021","unstructured":"Ank\u0131\u015fhan, H., & \u0130nam, S. C. (2021). Voice pathology detection by using the deep network architecture. Applied Soft Computing, 106, 107310. https:\/\/doi.org\/10.1016\/j.asoc.2021.107310","journal-title":"Applied Soft Computing"},{"issue":"4","key":"10120_CR6","doi-asserted-by":"publisher","first-page":"1219","DOI":"10.1044\/2014_JSLHR-S-12-0418","volume":"57","author":"L Bailly","year":"2014","unstructured":"Bailly, L., Bernardoni, N. H., M\u00fcller, F., Rohlfs, A. K., & Hess, M. (2014). Ventricular-fold dynamics in human phonation. Journal of Speech, Language, and Hearing Research, 57(4), 1219\u20131242. https:\/\/doi.org\/10.1044\/2014_JSLHR-S-12-0418","journal-title":"Journal of Speech, Language, and Hearing Research"},{"issue":"3","key":"10120_CR7","doi-asserted-by":"publisher","first-page":"403","DOI":"10.1067\/s0892-1997(03)00018-3","volume":"17","author":"A Behrman","year":"2003","unstructured":"Behrman, A., Dahl, L. D., Abramson, A. L., & Schutte, H. K. (2003). Anterior-posterior and medial compression of the supraglottis: Signs of nonorganic dysphonia or normal postures? Journal of Voice, 17(3), 403\u2013410. https:\/\/doi.org\/10.1067\/s0892-1997(03)00018-3","journal-title":"Journal of Voice"},{"key":"10120_CR8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.1979.1170788","author":"M Berouti","year":"1979","unstructured":"Berouti, M., Schwartz, R., & Makoul, J. (1979). Enhancement of speech corrupted by additive noise. IEEE Transactions on Acoustics, Speech, and Signal Processing. https:\/\/doi.org\/10.1109\/ICASSP.1979.1170788","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"issue":"4","key":"10120_CR9","first-page":"2340","volume":"4","author":"S Brijesh Anilbhai","year":"2017","unstructured":"Brijesh Anilbhai, S., & Kinnar, V. (2017). Spectral subtraction and MMSE: A hybrid approach for speech enhancement. International Research Journal of Engineering and Technology (IRJET), 4(4), 2340\u20132343.","journal-title":"International Research Journal of Engineering and Technology (IRJET)"},{"issue":"1","key":"10120_CR10","doi-asserted-by":"publisher","first-page":"44","DOI":"10.1016\/j.jvoice.2009.07.002","volume":"25","author":"M Brockmann","year":"2011","unstructured":"Brockmann, M., Drinnan, M. J., Storck, C., & Carding, P. N. (2011). Reliable jitter and shimmer measurements in voice clinics: The relevance of vowel, gender, vocal intensity, and fundamental frequency effects in a typical clinical task. Journal of Voice, 25(1), 44\u201353. https:\/\/doi.org\/10.1016\/j.jvoice.2009.07.002","journal-title":"Journal of Voice"},{"key":"10120_CR11","unstructured":"Brockmann-Bauser, M. (2012). Improving jitter and shimmer measurements in normal voices. Phd Thesis of Newcastle University. http:\/\/theses.ncl.ac.uk\/jspui\/handle\/10443\/1472."},{"issue":"2","key":"10120_CR12","doi-asserted-by":"publisher","first-page":"201","DOI":"10.1111\/coa.12765","volume":"42","author":"P Carding","year":"2016","unstructured":"Carding, P., Bos-Clark, M., Fu, S., Gillivan-Murphy, P., Jones, S. M., & Walton, C. (2016). Evaluating the efficacy of voice therapy for functional, organic, and neurological voice disorders. National Library of Medicine, 42(2), 201\u2013217. https:\/\/doi.org\/10.1111\/coa.12765","journal-title":"National Library of Medicine"},{"key":"10120_CR13","doi-asserted-by":"publisher","first-page":"463","DOI":"10.1016\/j.bbe.2022.03.002","volume":"42","author":"M Chaiani","year":"2022","unstructured":"Chaiani, M., Selouani, S. A., Boudraa, M., & Sidi Yakoub, M. (2022). Voice disorder classification using speech enhancement and deep learning models. Biocybernetics and Biomedical Engineering, 42, 463\u2013480. https:\/\/doi.org\/10.1016\/j.bbe.2022.03.002","journal-title":"Biocybernetics and Biomedical Engineering"},{"issue":"3","key":"10120_CR14","doi-asserted-by":"publisher","first-page":"312","DOI":"10.1002\/mdc3.12609","volume":"5","author":"DS Chung","year":"2018","unstructured":"Chung, D. S., Wettroth, C., Hallett, M., & Maurer, C. W. (2018). Functional speech and voice disorders: Case series and literature review. Movement Disorders Clinical Practices, 5(3), 312\u2013316. https:\/\/doi.org\/10.1002\/mdc3.12609","journal-title":"Movement Disorders Clinical Practices"},{"issue":"4","key":"10120_CR15","doi-asserted-by":"publisher","first-page":"357","DOI":"10.1109\/TASSP.1980.1163420","volume":"28","author":"SB Davis","year":"1980","unstructured":"Davis, S. B., & Mermelstein, P. (1980). Comparison of parametric representations for monosyllabic word recognition in continuously spokens entences. IEEE Transactions on Acoustics, Speech, and Signal Processing, 28(4), 357\u2013366. https:\/\/doi.org\/10.1109\/TASSP.1980.1163420","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"key":"10120_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvoice.2022.08.028","author":"F Deli","year":"2022","unstructured":"Deli, F., Xuehui, Z., Dandan, C., & Weiping, H. (2022). Pathological voice detection based on phase reconstitution and convolutional neural network. Journal of Voice. https:\/\/doi.org\/10.1016\/j.jvoice.2022.08.028","journal-title":"Journal of Voice"},{"key":"10120_CR17","unstructured":"Disordered Voice Database. (1994). Version 1.03 (CD-ROM), MEEI, Voice and Speech Lab, Kay Elemetrics Corp, Boston, MA, USA."},{"key":"10120_CR18","unstructured":"Duffy, J. R. (2019). Motor speech disorders: Substrates, differential diagnosis, and management, 4th Ed. Retrieved from https:\/\/shop.elsevier.com\/books\/motor-speech-disorders\/duffy\/978-0-323-53054-5."},{"key":"10120_CR19","doi-asserted-by":"publisher","first-page":"1280","DOI":"10.1134\/S1064226914110059","volume":"59","author":"IMM El Emary","year":"2014","unstructured":"El Emary, I. M. M., Fezari, M., & Amara, F. (2014). Towards developing a voice pathologies detection system. Journal of Communications Technology and Electronics, 59, 1280\u20131288. https:\/\/doi.org\/10.1134\/S1064226914110059","journal-title":"Journal of Communications Technology and Electronics"},{"issue":"2","key":"10120_CR20","doi-asserted-by":"publisher","first-page":"443","DOI":"10.1109\/TASSP.1985.1164550","volume":"33","author":"Y Ephraim","year":"1985","unstructured":"Ephraim, Y., & Malah, D. (1985). Speech enhancement using a minimum mean-square error log-spectral amplitude estimator. IEEE Transactions on Acoustics, Speech, and Signal Processing, 33(2), 443\u2013445. https:\/\/doi.org\/10.1109\/TASSP.1985.1164550","journal-title":"IEEE Transactions on Acoustics, Speech, and Signal Processing"},{"issue":"5","key":"10120_CR21","doi-asserted-by":"publisher","first-page":"643","DOI":"10.4218\/etrij.2017-0260","volume":"40","author":"A Farhadipour","year":"2018","unstructured":"Farhadipour, A., Veisi, H., Asgari, M., & Keyvanrad, M. A. (2018). Dysarthric speaker identification with different degrees of dysarthria severity using deep belief networks. ETRI Journal, 40(5), 643\u2013652. https:\/\/doi.org\/10.4218\/etrij.2017-0260","journal-title":"ETRI Journal"},{"key":"10120_CR22","doi-asserted-by":"publisher","unstructured":"Gholamalinezhad, H., & Khosravi, H. (2020). Pooling methods in deep neural networks, a review. https:\/\/doi.org\/10.48550\/arXiv.2009.07485.","DOI":"10.48550\/arXiv.2009.07485"},{"key":"10120_CR23","doi-asserted-by":"publisher","first-page":"662","DOI":"10.1016\/j.procs.2019.12.233","volume":"164","author":"V Guedes","year":"2019","unstructured":"Guedes, V., Teixeira, F., Oliveira, A., Fernandes, J., Silva, L., Junior, A., & Teixeira, J. P. (2019). Transfer learning with audioset to voice pathologies identification in continuous speech. Procedia Computer Science, 164, 662\u2013669. https:\/\/doi.org\/10.1016\/j.procs.2019.12.233","journal-title":"Procedia Computer Science"},{"key":"10120_CR24","doi-asserted-by":"publisher","unstructured":"Gupta, V. K., Bhowmick, A., Mahesh, C., & Saran, S. N. (2011). Speech enhancement using MMSE estimation and spectral subtraction methods. In International conference on devices and communications (ICDeCom) (pp. 1\u20135). https:\/\/doi.org\/10.1109\/ICDECOM.2011.5738532.","DOI":"10.1109\/ICDECOM.2011.5738532"},{"issue":"11","key":"10120_CR25","doi-asserted-by":"publisher","first-page":"82","DOI":"10.14569\/IJACSA.2018.091112","volume":"9","author":"R Hamdi","year":"2018","unstructured":"Hamdi, R., Hajji, S., & Cherif, A. (2018). Voice pathology recognition and classification using noise related features. International Journal of Advanced Computer Science and Applications (IJACSA), 9(11), 82\u201387. https:\/\/doi.org\/10.14569\/IJACSA.2018.091112","journal-title":"International Journal of Advanced Computer Science and Applications (IJACSA)"},{"key":"10120_CR26","doi-asserted-by":"publisher","unstructured":"Hara, K., Saito, D., Shouno, H. (2015). Analysis of function of rectified linear unit used in deep learning. In International joint conference on neural networks (IJCNN) (pp. 1\u20138). https:\/\/doi.org\/10.1109\/IJCNN.2015.7280578.","DOI":"10.1109\/IJCNN.2015.7280578"},{"key":"10120_CR27","doi-asserted-by":"publisher","unstructured":"Harar, P., Alonso-Hernandezy, J. B., Mekyska, J., Galaz, Z., Burget, R., & Smekal, Z. (2019). Voice pathology detection using deep learning: a preliminary study. In International conference and workshop on bioinspired intelligence (IWOBI) (pp. 1\u20134). https:\/\/doi.org\/10.1109\/IWOBI.2017.7985525.","DOI":"10.1109\/IWOBI.2017.7985525"},{"key":"10120_CR28","doi-asserted-by":"publisher","first-page":"7806","DOI":"10.1109\/ACCESS.2016.2626316","volume":"4","author":"MS Hossain","year":"2016","unstructured":"Hossain, M. S., & Muhammad, G. (2016). Healthcare big data voice pathology assessment framework. IEEE Access, 4, 7806\u20137815. https:\/\/doi.org\/10.1109\/ACCESS.2016.2626316","journal-title":"IEEE Access"},{"key":"10120_CR29","doi-asserted-by":"publisher","unstructured":"Janbakhshi, P., Kodrasi. I. (2022a). Adversarial-free speaker identity-invariant representation learning for automatic dysarthric speech classification. In Proceedings of the annual conference of the international speech communication (Interspeech) (pp. 2138\u20132142). https:\/\/doi.org\/10.21437\/Interspeech.2022-402.","DOI":"10.21437\/Interspeech.2022-402"},{"key":"10120_CR30","doi-asserted-by":"publisher","unstructured":"Janbakhshi, P., Kodrasi. (2022b). Experimental investigation on stft phase representations for deep learning-based dysarthric speech detection. In Proceedings of the IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp. 6477\u20136481). https:\/\/doi.org\/10.48550\/arXiv.2110.03283.","DOI":"10.48550\/arXiv.2110.03283"},{"key":"10120_CR31","volume-title":"Dysarthria. StatPearls [Internet]","author":"DK Jayaraman","year":"2023","unstructured":"Jayaraman, D. K., & Das, J. M. (2023). Dysarthria. StatPearls [Internet]. StatPearls Publishing."},{"key":"10120_CR32","doi-asserted-by":"publisher","unstructured":"Joshy, A. A., & Rajan, R. (2021). Automated dysarthria severity classification using deep learning frameworks. In 28th European signal processing conference (EUSIPCO) (pp. 116\u2013120). https:\/\/doi.org\/10.23919\/Eusipco47968.2020.9287741.","DOI":"10.23919\/Eusipco47968.2020.9287741"},{"key":"10120_CR33","doi-asserted-by":"publisher","first-page":"233","DOI":"10.1016\/j.bbe.2015.11.004","volume":"36","author":"KL Kadi","year":"2016","unstructured":"Kadi, K. L., Selouani, S. A., Boudraa, B., & Boudraa, M. (2016). Fully automated speaker identification and intelligibility assessment in dysarthria disease using auditory knowledge. Biocybernetics and Biomedical Engineering, 36, 233\u2013247. https:\/\/doi.org\/10.1016\/j.bbe.2015.11.004","journal-title":"Biocybernetics and Biomedical Engineering"},{"issue":"13","key":"10120_CR34","doi-asserted-by":"publisher","first-page":"45","DOI":"10.5120\/16858-6739","volume":"96","author":"N Kaladharan","year":"2014","unstructured":"Kaladharan, N. (2014). Speech enhancement by spectral subtraction method. International Journal of Computer Applications, 96(13), 45\u201348. https:\/\/doi.org\/10.5120\/16858-6739","journal-title":"International Journal of Computer Applications"},{"issue":"6","key":"10120_CR35","doi-asserted-by":"publisher","first-page":"420","DOI":"10.1097\/MOO.0b013e328331a7f8","volume":"17","author":"PD Karkos","year":"2009","unstructured":"Karkos, P. D., & McCormick, M. (2009). The etiology of vocal fold nodules in adults. Current Opinion in Otolaryngology & Head and Neck Surgery, 17(6), 420\u2013423. https:\/\/doi.org\/10.1097\/MOO.0b013e328331a7f8","journal-title":"Current Opinion in Otolaryngology & Head and Neck Surgery"},{"key":"10120_CR36","doi-asserted-by":"publisher","DOI":"10.1002\/9781444301007.ch22","author":"RD Kent","year":"2008","unstructured":"Kent, R. D., & Kim, Y. (2008). Acoustic analysis of speech. In The handbook of clinical linguistics(pp. 360\u2013380). https:\/\/doi.org\/10.1002\/9781444301007.ch22","journal-title":"Acoustic Analysis of Speech."},{"issue":"7","key":"10120_CR37","doi-asserted-by":"publisher","first-page":"1315","DOI":"10.1109\/TASLP.2016.2545928","volume":"24","author":"C Kim","year":"2016","unstructured":"Kim, C., & Stern, R. M. (2016). Power-normalized cepstral coefficients (PNCC) for robust speech recognition. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 24(7), 1315\u20131329. https:\/\/doi.org\/10.1109\/TASLP.2016.2545928","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10120_CR38","unstructured":"Kishore, P. (2011). Speech technology: A practical introduction, topic: spectrogram, cepstrum, and mel frequency analysis. Retrieved from https:\/\/www.cs.brandeis.edu\/~cs136a\/CS136a_docs\/KishorePrahallad_CMU_mfcc.pdf."},{"key":"10120_CR39","doi-asserted-by":"publisher","unstructured":"Klambauer, G., Unterthiner, T., Mayr, A., & Hochreiter, S. (2017). Self-normalizing neural networks. In 31st conference on neural information processing systems (NIPS) (pp. 972\u2013981). https:\/\/doi.org\/10.48550\/arXiv.1706.02515.","DOI":"10.48550\/arXiv.1706.02515"},{"issue":"4","key":"10120_CR40","doi-asserted-by":"publisher","first-page":"3204","DOI":"10.3390\/su15043204","volume":"15","author":"A Ksibi","year":"2023","unstructured":"Ksibi, A., Hakami, N. A., Alturki, N., Asiri, M. M., Zakariah, M., & Ayadi, M. (2023). Voice pathology detection using a two-level classifier based on combined CNN\u2013RNN architecture. Sustainability, 15(4), 3204. https:\/\/doi.org\/10.3390\/su15043204","journal-title":"Sustainability"},{"issue":"14","key":"10120_CR41","doi-asserted-by":"publisher","first-page":"23","DOI":"10.5120\/ijca2016909507","volume":"139","author":"DS Kulkarni","year":"2016","unstructured":"Kulkarni, D. S., Deshmukh, R. R., & Shrishrimal, P. (2016). A review of speech signal enhancement techniques. International Journal of Computer Applications, 139(14), 23\u201326. https:\/\/doi.org\/10.5120\/ijca2016909507","journal-title":"International Journal of Computer Applications"},{"key":"10120_CR42","doi-asserted-by":"publisher","unstructured":"Lee, M. (2023). GELU activation function in deep learning: A comprehensive mathematical analysis and performance. https:\/\/doi.org\/10.48550\/arXiv.2305.12073.","DOI":"10.48550\/arXiv.2305.12073"},{"key":"10120_CR43","doi-asserted-by":"publisher","first-page":"197","DOI":"10.1109\/PROC.1979.11540","volume":"12","author":"JS Lim","year":"1979","unstructured":"Lim, J. S., & Oppenheim, A. V. (1979). Enhancement and bandwidth compression of noisy speech. Proceedings of the IEEE, 12, 197\u2013210. https:\/\/doi.org\/10.1109\/PROC.1979.11540","journal-title":"Proceedings of the IEEE"},{"key":"10120_CR44","doi-asserted-by":"publisher","DOI":"10.1201\/9781420015836","volume-title":"Speech enhancement: Theory and practice","author":"PC Loizou","year":"2007","unstructured":"Loizou, P. C. (2007). Speech enhancement: Theory and practice. CRC Press. https:\/\/doi.org\/10.1201\/9781420015836"},{"key":"10120_CR45","doi-asserted-by":"publisher","unstructured":"Mayle, A., Mou, Z., Bunescu, R., Mirshekarian, S., Xu, L., & Liu, C. (2019). Diagnosing dysarthria with long short-term memory networks. In Proceedings of the annual conference of the international speech communication (Interspeech) (pp. 4514\u20134518). https:\/\/doi.org\/10.21437\/Interspeech.2019-2903.","DOI":"10.21437\/Interspeech.2019-2903"},{"key":"10120_CR46","doi-asserted-by":"publisher","unstructured":"Mediratta, I., Saha, S., Mathur, S. (2021). LipARELU: ARELU networks aided by Lipschitz Acceleration. In International joint conference on neural networks (IJCNN) (pp. 1\u20138). https:\/\/doi.org\/10.1109\/IJCNN52387.2021.9533853.","DOI":"10.1109\/IJCNN52387.2021.9533853"},{"key":"10120_CR47","doi-asserted-by":"publisher","first-page":"119790","DOI":"10.1016\/j.eswa.2023.119790","volume":"223","author":"HMA Mohammed","year":"2023","unstructured":"Mohammed, H. M. A., Omergolu, A. N., & Oral, E. A. (2023). MMHFNet: Multi-modal and multi-layer hybrid fusion network for voice pathology detection. Expert Systems and Applications, 223, 119790. https:\/\/doi.org\/10.1016\/j.eswa.2023.119790","journal-title":"Expert Systems and Applications"},{"key":"10120_CR48","doi-asserted-by":"publisher","first-page":"1925","DOI":"10.1109\/TASLP.2021.3078364","volume":"29","author":"NP Narendra","year":"2021","unstructured":"Narendra, N. P., Schuller, B., & Alku, P. (2021). The detection of Parkinson\u2019s disease from speech using voice source information. IEEE\/ACM Transactions on Audio, Speech, and Language Processing, 29, 1925\u20131936. https:\/\/doi.org\/10.1109\/TASLP.2021.3078364","journal-title":"IEEE\/ACM Transactions on Audio, Speech, and Language Processing"},{"key":"10120_CR49","doi-asserted-by":"publisher","first-page":"7264","DOI":"10.1038\/s41598-023-34461-9","volume":"13","author":"X Peng","year":"2023","unstructured":"Peng, X., Xu, H., Liu, J., Wang, J., & He, C. (2023). Voice disorder classification using convolutional neural network based on deep transfer learning. Scientific Reports, 13, 7264. https:\/\/doi.org\/10.1038\/s41598-023-34461-9","journal-title":"Scientific Reports"},{"issue":"9","key":"10120_CR50","doi-asserted-by":"publisher","first-page":"1215","DOI":"10.1109\/5.237532","volume":"81","author":"JW Picone","year":"1993","unstructured":"Picone, J. W. (1993). Signal modeling techniques in speech recognition. Proceedings of the IEEE, 81(9), 1215\u20131247. https:\/\/doi.org\/10.1109\/5.237532","journal-title":"Proceedings of the IEEE"},{"key":"10120_CR51","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2007-386","author":"G Pouchoulin","year":"2007","unstructured":"Pouchoulin, G., Fredouille, C., Bonastre, J. F., Ghio, A., & Giovanni, A. (2007). Frequency study for the characterization of the dysphonic voices. In\u00a0Interspeech. https:\/\/doi.org\/10.21437\/Interspeech.2007-386","journal-title":"Interspeech"},{"key":"10120_CR52","unstructured":"P\u00fctzer, M., & Barry, W. J. (2007). Saarbruecken Voice Database. Institut f\u00fcr Phonetik. Universit\u00e4t des Saarlandes. Retrieved from https:\/\/stimmdb.coli.uni-saarland.de\/help_en.php4."},{"key":"10120_CR53","doi-asserted-by":"publisher","first-page":"521","DOI":"10.48550\/arXiv.2306.00689","volume":"26","author":"AS Shakeel","year":"2023","unstructured":"Shakeel, A. S., Sahidullah, M. D., Fabrice, H., & Slim, O. (2023). Stuttering detection using speaker representations and self-supervised contextual embeddings. International Journal of Speech Technology, 26, 521\u2013530. https:\/\/doi.org\/10.48550\/arXiv.2306.00689","journal-title":"International Journal of Speech Technology"},{"key":"10120_CR54","doi-asserted-by":"publisher","unstructured":"Shakeel, A. S., Sahidullah, M. D., Fabrice, H., & Slim, O. (2021). StutterNet: Stuttering detection using time delay neural network. In 29th European signal processing conference (EUSIPCO) (pp. 426\u2013430). https:\/\/doi.org\/10.48550\/arXiv.2105.05599.","DOI":"10.48550\/arXiv.2105.05599"},{"key":"10120_CR55","doi-asserted-by":"publisher","unstructured":"Souissi, N., & Cherif, A. (2015). Dimensionality reduction for voice disorders identification system based on mel frequency cepstral coefficients and support vector machine. In 7th international conference on modelling, identification and control (ICMIC) (pp. 1\u20136). https:\/\/doi.org\/10.1109\/ICMIC.2015.7409479.","DOI":"10.1109\/ICMIC.2015.7409479"},{"key":"10120_CR56","doi-asserted-by":"publisher","first-page":"107854","DOI":"10.1016\/j.apacoust.2020.107854","volume":"177","author":"S Souli","year":"2021","unstructured":"Souli, S., Amami, R., & Ben Yahia, S. (2021). A robust pathological voices recognition system based on DCNN and scattering transform. Applied Acoustics, 177, 107854. https:\/\/doi.org\/10.1016\/j.apacoust.2020.107854","journal-title":"Applied Acoustics"},{"key":"10120_CR57","doi-asserted-by":"publisher","unstructured":"Staudemeyer, R. C., & Morris, E. R. (2019). Understanding LSTM\u2014a tutorial into long short-term memory recurrent neural networks. https:\/\/doi.org\/10.48550\/arXiv.1909.09586.","DOI":"10.48550\/arXiv.1909.09586"},{"key":"10120_CR58","unstructured":"Strand, O. M., & Egeberg, A. (2004). Cepstral mean and variance normalization in the model domain. In Proceedings of the COST\/ISCA tutorial and research workshop on robustness issues in conversational interaction, paper 38."},{"key":"10120_CR59","first-page":"365","volume":"30","author":"K Sumin","year":"2021","unstructured":"Sumin, K., Chung, W., & Lee, J. (2021). Acoustic full waveform inversion using discrete cosine transform (DCT). Journal of Seismic Exploration, 30, 365\u2013380.","journal-title":"Journal of Seismic Exploration"},{"key":"10120_CR60","doi-asserted-by":"publisher","unstructured":"Suresh, M., & Thomas, J. (2023). Review on dysarthric speech severity level classification frameworks. In International conference on control, communication and computing (ICCC). https:\/\/doi.org\/10.1109\/ICCC57789.2023.10165636.","DOI":"10.1109\/ICCC57789.2023.10165636"},{"issue":"5","key":"10120_CR61","doi-asserted-by":"publisher","first-page":"1112","DOI":"10.1016\/j.protcy.2013.12.124","volume":"9","author":"JP Teixeira","year":"2013","unstructured":"Teixeira, J. P., Oliveira, C., & Lopes, C. (2013). Vocal acoustic analysis jitter, shimmer and hnr parameters. Procedia Technology, 9(5), 1112\u20131122. https:\/\/doi.org\/10.1016\/j.protcy.2013.12.124","journal-title":"Procedia Technology"},{"issue":"1","key":"10120_CR62","doi-asserted-by":"publisher","first-page":"47","DOI":"10.5681\/jcvtr.2014.009","volume":"6","author":"SJS Toutounchi","year":"2014","unstructured":"Toutounchi, S. J. S., Eydi, M., Ej Golzari, S., Ghaffari, M. R., & Parvizian, N. (2014). Vocal cord paralysis and its etiologies: A prospective study. Journal of Cardiovascular and Thoracic Research, 6(1), 47\u201350. https:\/\/doi.org\/10.5681\/jcvtr.2014.009","journal-title":"Journal of Cardiovascular and Thoracic Research"},{"key":"10120_CR63","doi-asserted-by":"publisher","unstructured":"Vaiciukynas, E., Gelzinis, A., Verikas, A., & Bacauskiene, M. (2018). Parkinson\u2019s disease detection from speech using convolutional neural networks. In Smart objects and technologies for social good: Third international conference, (Vol. 233, pp. 206\u2013215). https:\/\/doi.org\/10.1007\/978-3-319-76111-4_21","DOI":"10.1007\/978-3-319-76111-4_21"},{"issue":"8","key":"10120_CR64","doi-asserted-by":"publisher","first-page":"1900","DOI":"10.1111\/j.1572-0241.2006.00630.x","volume":"101","author":"N Vakil","year":"2006","unstructured":"Vakil, N., van Zanten, S. V., Kahrilas, P., Dent, J., & Jones, R. (2006). The Montreal definition and classification of gastroesophageal reflux disease: A global evidence-based consensus. The American Journal of Gastroenterology, 101(8), 1900\u20131920. https:\/\/doi.org\/10.1111\/j.1572-0241.2006.00630.x","journal-title":"The American Journal of Gastroenterology"},{"key":"10120_CR65","doi-asserted-by":"publisher","unstructured":"V\u00e1squez-Correa, J. C., Orozco-Arroyave, J. R., & N\u00f6th, E. (2017). Convolutional neural network to model articulation impairments in patients with Parkinson\u2019s disease. In Proceedings of the annual conference of the international speech communication (Interspeech) (pp. 314\u2013318). https:\/\/doi.org\/10.21437\/Interspeech.2017-1078.","DOI":"10.21437\/Interspeech.2017-1078"},{"key":"10120_CR66","doi-asserted-by":"publisher","first-page":"25","DOI":"10.1109\/OJEMB.2022.3151233","volume":"3","author":"SS Wang","year":"2022","unstructured":"Wang, S. S., Wang, C. T., Lai, C. C., Tsao, Y., & Fang, S. H. (2022). Continuous speech for improved learning pathological voice disorders. IEEE Open Journal of Engineering in Medicine and Biology, 3, 25\u201333. https:\/\/doi.org\/10.1109\/OJEMB.2022.3151233","journal-title":"IEEE Open Journal of Engineering in Medicine and Biology"},{"issue":"5","key":"10120_CR67","doi-asserted-by":"publisher","first-page":"582","DOI":"10.1016\/s1808-8694(15)31261-1","volume":"71","author":"HF Westzner","year":"2005","unstructured":"Westzner, H. F., Schreiber, S., & Amaro, L. (2005). Analysis of fundamental frequency, jitter, shimmer and vocal intensity in children with phonological disorders. Brazilian Journal of Orthinolaryngology, 71(5), 582\u2013588. https:\/\/doi.org\/10.1016\/s1808-8694(15)31261-1","journal-title":"Brazilian Journal of Orthinolaryngology"},{"key":"10120_CR68","doi-asserted-by":"publisher","unstructured":"Wu, H., Soraghan, J., Lowit, A., & Di-Caterina, G. (2018). A deep learning method for pathological voice detection using convolutional deep belief networks. In Proceedings of the annual conference of the international speech communication (Interspeech) (pp. 446\u2013450). https:\/\/doi.org\/10.21437\/Interspeech.2018-1351.","DOI":"10.21437\/Interspeech.2018-1351"},{"key":"10120_CR69","unstructured":"Xiaoyu, L. (2018). Deep convolutional and LSTM neural networks for acoustic modelling in automatic speech recognition. Retrieved from https:\/\/cs231n.stanford.edu\/reports\/2017\/pdfs\/804.pdf."},{"key":"10120_CR70","unstructured":"Xing Luo, O. (2019). Deep learning for speech enhancement- a study on WaveNet, GANs and general RNN architectures. Retrieved from http:\/\/www.divaportal.org\/smash\/get\/diva2:1355369\/FULLTEXT01.pdf."},{"key":"10120_CR71","doi-asserted-by":"publisher","DOI":"10.2478\/sjph-2018-0003","author":"M Zabret","year":"2018","unstructured":"Zabret, M., Ho\u010devar Bolte\u017ear, I., & \u0160ereg Bahar, M. (2018). The importance of the occupational vocal load for the occurrence and treatment of organic voice disorders. National Library of Medicine. https:\/\/doi.org\/10.2478\/sjph-2018-0003","journal-title":"National Library of Medicine"},{"issue":"4","key":"10120_CR72","doi-asserted-by":"publisher","first-page":"2614","DOI":"10.1121\/1.4964509","volume":"140","author":"Z Zhaoyan","year":"2016","unstructured":"Zhaoyan, Z. (2016). Mechanics of human voice production and control. The Journal of Acoustical Society of America, 140(4), 2614\u20132635. https:\/\/doi.org\/10.1121\/1.4964509","journal-title":"The Journal of Acoustical Society of America"},{"issue":"1","key":"10120_CR73","doi-asserted-by":"publisher","first-page":"108417","DOI":"10.1016\/j.apacoust.2021.108417","volume":"185","author":"C Zhou","year":"2022","unstructured":"Zhou, C., Wu, Y., Fan, Z., Zhang, X., Wu, D., & Tao, Z. (2022). Gammatone spectral latitude features extraction for pathological voice detection and classification. Applied Acoustics, 185(1), 108417. https:\/\/doi.org\/10.1016\/j.apacoust.2021.108417","journal-title":"Applied Acoustics"},{"key":"10120_CR74","doi-asserted-by":"publisher","first-page":"698","DOI":"10.1016\/j.jvoice.2015.08.013","volume":"30","author":"P Zhuge","year":"2016","unstructured":"Zhuge, P., You, H., Wang, H., Zhang, Y., & Du, H. (2016). An analysis of the effects of voice therapy on patients with early vocal fold polyps. Journal of Voice, 30, 698\u2013704. https:\/\/doi.org\/10.1016\/j.jvoice.2015.08.013","journal-title":"Journal of Voice"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10120-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10120-w\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10120-w.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,7,22]],"date-time":"2024-07-22T16:10:26Z","timestamp":1721664626000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10120-w"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,6]]},"references-count":74,"journal-issue":{"issue":"2","published-print":{"date-parts":[[2024,6]]}},"alternative-id":["10120"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10120-w","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,6]]},"assertion":[{"value":"17 November 2023","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"21 June 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"5 July 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}