{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,3]],"date-time":"2026-07-03T10:35:19Z","timestamp":1783074919477,"version":"3.54.6"},"reference-count":44,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2024,8,10]],"date-time":"2024-08-10T00:00:00Z","timestamp":1723248000000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,8,10]],"date-time":"2024-08-10T00:00:00Z","timestamp":1723248000000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,9]]},"DOI":"10.1007\/s10772-024-10134-4","type":"journal-article","created":{"date-parts":[[2024,8,10]],"date-time":"2024-08-10T07:02:32Z","timestamp":1723273352000},"page":"739-751","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":15,"title":["Voice pathology detection on spontaneous speech data using deep learning models"],"prefix":"10.1007","volume":"27","author":[{"given":"Sahar","family":"Farazi","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-6733-3702","authenticated-orcid":false,"given":"Yasser","family":"Shekofteh","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,8,10]]},"reference":[{"key":"10134_CR1","doi-asserted-by":"publisher","first-page":"122136","DOI":"10.1109\/ACCESS.2022.3223444","volume":"10","author":"ZK Abdul","year":"2022","unstructured":"Abdul, Z. K., & Al-Talabani, A. K. (2022). Mel frequency cepstral coefficient and its applications: A review. IEEE Access, 10, 122136\u2013122158.","journal-title":"IEEE Access"},{"issue":"1","key":"10134_CR2","doi-asserted-by":"publisher","first-page":"855","DOI":"10.1515\/jisys-2022-0058","volume":"31","author":"NQ Abdulmajeed","year":"2022","unstructured":"Abdulmajeed, N. Q., Al-Khateeb, B., & Mohammed, M. A. (2022). A review on voice pathology: Taxonomy, diagnosis, medical procedures and detection techniques, open challenges, limitations, and recommendations for future directions. Journal of Intelligent Systems, 31(1), 855\u2013875.","journal-title":"Journal of Intelligent Systems"},{"key":"10134_CR3","doi-asserted-by":"publisher","DOI":"10.1111\/exsy.13327","author":"NQ Abdulmajeed","year":"2023","unstructured":"Abdulmajeed, N. Q., Al-Khateeb, B., & Mohammed, M. A. (2023). Voice pathology identification system using a deep learning approach based on unique feature selection sets. Expert Systems. https:\/\/doi.org\/10.1111\/exsy.13327","journal-title":"Expert Systems"},{"key":"10134_CR4","doi-asserted-by":"crossref","unstructured":"Ali, Z., Alsulaiman, M., Muhammad, G., Elamvazuthi, I., & Mesallam, T. A. (2013). Vocal fold disorder detection based on continuous speech by using MFCC and GMM. In 2013 7th IEEE GCC conference and exhibition (GCC). IEEE.","DOI":"10.1109\/IEEEGCC.2013.6705792"},{"issue":"6","key":"10134_CR5","doi-asserted-by":"publisher","first-page":"757","DOI":"10.1016\/j.jvoice.2015.08.010","volume":"30","author":"Z Ali","year":"2016","unstructured":"Ali, Z., Elamvazuthi, I., Alsulaiman, M., & Muhammad, G. (2016). Automatic voice pathology detection with running speech by using estimation of auditory spectrum and cepstral coefficients based on the all-pole model. Journal of Voice, 30(6), 757.","journal-title":"Journal of Voice"},{"key":"10134_CR6","doi-asserted-by":"publisher","first-page":"19","DOI":"10.1016\/j.future.2018.02.021","volume":"85","author":"Z Ali","year":"2018","unstructured":"Ali, Z., Hossain, M. S., Muhammad, G., & Sangaiah, A. K. (2018). An intelligent healthcare system for detection and classification to discriminate vocal fold disorders. Future Generation Computer Systems, 85, 19\u201328.","journal-title":"Future Generation Computer Systems"},{"key":"10134_CR7","doi-asserted-by":"publisher","first-page":"3900","DOI":"10.1109\/ACCESS.2017.2680467","volume":"5","author":"Z Ali","year":"2017","unstructured":"Ali, Z., Muhammad, G., & Alhamid, M. F. (2017). An automatic health monitoring system for patients suffering from voice complications in smart cities. IEEE Access, 5, 3900\u20133908.","journal-title":"Ieee Access"},{"key":"10134_CR8","doi-asserted-by":"crossref","unstructured":"Al-Sabaawi, A., Ibrahim, H. M., Arkah, Z. M., Al-Amidie, M., & Alzubaidi, L. (2020). Amended convolutional neural network with global average pooling for image classification. In International conference on intelligent systems design and applications. Springer.","DOI":"10.1007\/978-3-030-71187-0_16"},{"key":"10134_CR9","doi-asserted-by":"publisher","DOI":"10.1016\/j.asoc.2021.107310","volume":"106","author":"H Ank\u0131\u015fhan","year":"2021","unstructured":"Ank\u0131\u015fhan, H., & \u0130nam, S. \u00c7. (2021). Voice pathology detection by using the deep network architecture. Applied Soft Computing, 106, 107310.","journal-title":"Applied Soft Computing"},{"key":"10134_CR10","volume-title":"Deep learning","author":"Y Bengio","year":"2017","unstructured":"Bengio, Y., Goodfellow, I., & Courville, A. (2017). Deep learning. MIT Press Cambridge."},{"key":"10134_CR11","volume-title":"Automatic detection of alzheimer\u2019s disease using spontaneous speech only","author":"J Chen","year":"2021","unstructured":"Chen, J., Ye, J., Tang, F., & Zhou, J. (2021). Automatic detection of Alzheimer\u2019s disease using spontaneous speech only. NIH Public Access."},{"issue":"2","key":"10134_CR12","doi-asserted-by":"publisher","first-page":"288","DOI":"10.1016\/j.jvoice.2020.05.029","volume":"36","author":"L Chen","year":"2022","unstructured":"Chen, L., & Chen, J. (2022). Deep neural network for automatic classification of pathological voice signals. Journal of Voice, 36(2), 288.","journal-title":"Journal of Voice"},{"key":"10134_CR13","doi-asserted-by":"crossref","unstructured":"Chuang, Z.-Y., Yu, X.-T., Chen, J.-Y., Hsu, Y.-T., Xu, Z.-Z., Wang, C.-T., Lin, F.-C., & Fang, S.-H. (2018). Dnn-based approach to detect and classify pathological voice. In 2018 IEEE international conference on big data (Big Data). IEEE.","DOI":"10.1109\/BigData.2018.8622317"},{"issue":"6","key":"10134_CR14","doi-asserted-by":"publisher","first-page":"1451","DOI":"10.1007\/s12559-020-09813-6","volume":"13","author":"G Chugh","year":"2021","unstructured":"Chugh, G., Kumar, S., & Singh, N. (2021). Survey on machine learning and deep learning applications in breast cancer diagnosis. Cognitive Computation, 13(6), 1451\u20131470.","journal-title":"Cognitive Computation"},{"key":"10134_CR15","first-page":"1565","volume":"24","author":"P Deepa","year":"2022","unstructured":"Deepa, P., & Khilar, R. (2022). Speech technology in healthcare. Measurement: Sensors, 24, 1565.","journal-title":"Measurement: Sensors"},{"key":"10134_CR16","unstructured":"Association, A. S.-L.-H. (2009). Consensus Auditory-Perceptual Evaluation of Voice (CAPE-V) ASHA Special Interest Group 3, Voice and Voice Disorders. American Speech-Language-Hearing Association."},{"key":"10134_CR17","doi-asserted-by":"publisher","DOI":"10.1007\/11550907_126","volume-title":"Bidirectional LSTM networks for improved phoneme classification and recognition","author":"A Graves","year":"2005","unstructured":"Graves, A., Fern\u00e1ndez, S., & Schmidhuber, J. (2005). Bidirectional LSTM networks for improved phoneme classification and recognition. Springer."},{"issue":"5\u20136","key":"10134_CR18","doi-asserted-by":"publisher","first-page":"602","DOI":"10.1016\/j.neunet.2005.06.042","volume":"18","author":"A Graves","year":"2005","unstructured":"Graves, A., & Schmidhuber, J. (2005). Framewise phoneme classification with bidirectional LSTM and other neural network architectures. Neural Networks, 18(5\u20136), 602\u2013610.","journal-title":"Neural Networks"},{"key":"10134_CR19","doi-asserted-by":"publisher","first-page":"354","DOI":"10.1016\/j.patcog.2017.10.013","volume":"77","author":"J Gu","year":"2018","unstructured":"Gu, J., Wang, Z., Kuen, J., Ma, L., Shahroudy, A., Shuai, B., Liu, T., Wang, X., Wang, G., & Cai, J. (2018). Recent advances in convolutional neural networks. Pattern Recognition, 77, 354\u2013377.","journal-title":"Pattern Recognition"},{"issue":"6","key":"10134_CR20","doi-asserted-by":"publisher","first-page":"947","DOI":"10.1016\/j.jvoice.2018.07.014","volume":"33","author":"S Hegde","year":"2019","unstructured":"Hegde, S., Shetty, S., Rai, S., & Dodderi, T. (2019). A survey on machine learning approaches for automatic detection of voice disorders. Journal of Voice, 33(6), 947.","journal-title":"Journal of Voice"},{"key":"10134_CR21","doi-asserted-by":"publisher","first-page":"66749","DOI":"10.1109\/ACCESS.2020.2985280","volume":"8","author":"R Islam","year":"2020","unstructured":"Islam, R., Tarique, M., & Abdel-Raheem, E. (2020). A survey on signal processing based pathological voice detection techniques. IEEE Access, 8, 66749\u201366776.","journal-title":"IEEE Access"},{"key":"10134_CR22","doi-asserted-by":"crossref","unstructured":"Jesus, L. M., Barney, A., Santos, R., Caetano, J., Jorge, J., & Couto, P. S. (2009). Universidade de Aveiro's voice evaluation protocol. In Tenth annual conference of the international speech communication association (Interspeech).","DOI":"10.21437\/Interspeech.2009-289"},{"key":"10134_CR23","doi-asserted-by":"crossref","unstructured":"Jesus, L. M., Belo, I., Machado, J., & Hall, A. (2017). The advanced voice function assessment databases (AVFAD): Tools for voice clinicians and speech research. Advances in Speech-Language Pathology.","DOI":"10.5772\/intechopen.69643"},{"key":"10134_CR24","volume-title":"The MIT encyclopedia of communication disorders","author":"RD Kent","year":"2004","unstructured":"Kent, R. D. (2004). The MIT encyclopedia of communication disorders. MIT Press."},{"issue":"4","key":"10134_CR25","doi-asserted-by":"publisher","first-page":"3204","DOI":"10.3390\/su15043204","volume":"15","author":"A Ksibi","year":"2023","unstructured":"Ksibi, A., Hakami, N. A., Alturki, N., Asiri, M. M., Zakariah, M., & Ayadi, M. (2023). Voice pathology detection using a two-level classifier based on combined CNN\u2013RNN architecture. Sustainability, 15(4), 3204.","journal-title":"Sustainability"},{"key":"10134_CR26","doi-asserted-by":"publisher","first-page":"342","DOI":"10.1109\/RBME.2020.3006860","volume":"14","author":"S Latif","year":"2020","unstructured":"Latif, S., Qadir, J., Qayyum, A., Usama, M., & Younis, S. (2020). Speech technology for healthcare: Opportunities, challenges, and state of the art. IEEE Reviews in Biomedical Engineering, 14, 342\u2013356.","journal-title":"IEEE Reviews in Biomedical Engineering"},{"issue":"15","key":"10134_CR27","doi-asserted-by":"publisher","first-page":"7149","DOI":"10.3390\/app11157149","volume":"11","author":"J-Y Lee","year":"2021","unstructured":"Lee, J.-Y. (2021). Experimental evaluation of deep learning methods for an intelligent pathological voice detection system using the saarbruecken voice database. Applied Sciences, 11(15), 7149.","journal-title":"Applied Sciences"},{"issue":"12","key":"10134_CR28","doi-asserted-by":"publisher","first-page":"6999","DOI":"10.1109\/TNNLS.2021.3084827","volume":"33","author":"Z Li","year":"2021","unstructured":"Li, Z., Liu, F., Yang, W., Peng, S., & Zhou, J. (2021). A survey of convolutional neural networks: Analysis, applications, and prospects. IEEE Transactions on Neural Networks and Learning Systems, 33(12), 6999\u20137019.","journal-title":"IEEE Transactions on Neural Networks and Learning Systems"},{"issue":"1","key":"10134_CR29","first-page":"878351","volume":"2017","author":"TA Mesallam","year":"2017","unstructured":"Mesallam, T. A., Farahat, M., Malki, K. H., Alsulaiman, M., Ali, Z., Al-Nasheri, A., & Muhammad, G. (2017). Development of the Arabic voice pathology database and its evaluation by using speech features and machine learning algorithms. Journal of Healthcare Engineering, 2017(1), 878351.","journal-title":"Journal of Healthcare Engineering"},{"issue":"11","key":"10134_CR30","doi-asserted-by":"publisher","first-page":"3723","DOI":"10.3390\/app10113723","volume":"10","author":"MA Mohammed","year":"2020","unstructured":"Mohammed, M. A., Abdulkareem, K. H., Mostafa, S. A., Khanapi Abd Ghani, M., Maashi, M. S., Garcia-Zapirain, B., Oleagordia, I., Alhakami, H., & Al-Dhief, F. T. (2020). Voice pathology detection and classification using convolutional neural network model. Applied Sciences, 10(11), 3723.","journal-title":"Applied Sciences"},{"key":"10134_CR31","doi-asserted-by":"publisher","first-page":"89198","DOI":"10.1109\/ACCESS.2021.3090317","volume":"9","author":"G Muhammad","year":"2021","unstructured":"Muhammad, G., & Alhussein, M. (2021). Convergence of artificial intelligence and internet of things in smart healthcare: A case study of voice pathology detection. IEEE Access, 9, 89198\u201389209.","journal-title":"IEEE Access"},{"key":"10134_CR32","doi-asserted-by":"publisher","first-page":"67745","DOI":"10.1109\/ACCESS.2020.2986171","volume":"8","author":"N Narendra","year":"2020","unstructured":"Narendra, N., & Alku, P. (2020). Glottal source information for pathological voice detection. IEEE Access, 8, 67745\u201367755.","journal-title":"IEEE Access"},{"key":"10134_CR33","doi-asserted-by":"crossref","unstructured":"Oliveira, B. F., Magalh\u00e3es, D. M., Ferreira, D. S., & Medeiros, F. N. (2020). Combined sustained vowels improve the performance of the Haar wavelet for pathological voice characterization. In 2020 International conference on systems, signals and image processing (IWSSIP), IEEE.","DOI":"10.1109\/IWSSIP48289.2020.9145258"},{"key":"10134_CR34","doi-asserted-by":"publisher","DOI":"10.1016\/j.jvoice.2022.02.009","author":"CL Payten","year":"2022","unstructured":"Payten, C. L., Chiapello, G., Weir, K. A., & Madill, C. J. (2022). Frameworks, terminology and definitions used for the classification of voice disorders: A scoping review. Journal of Voice. https:\/\/doi.org\/10.1016\/j.jvoice.2022.02.009","journal-title":"Journal of Voice"},{"key":"10134_CR35","doi-asserted-by":"crossref","unstructured":"Ribas, D., Miguel, A., Ortega, A., & Lleida, E. (2023a). On the problem of data availability in automatic voice disorder detection. In HEALTHINF, (pp. 330\u2013337).","DOI":"10.5220\/0011669300003414"},{"key":"10134_CR36","doi-asserted-by":"publisher","first-page":"14915","DOI":"10.1109\/ACCESS.2023.3243986","volume":"11","author":"D Ribas","year":"2023","unstructured":"Ribas, D., Pastor, M. A., Miguel, A., Mart\u00ednez, D., Ortega, A., & Lleida, E. (2023b). Automatic voice disorder detection using self-supervised representations. IEEE Access, 11, 14915\u201314927.","journal-title":"IEEE Access"},{"issue":"6","key":"10134_CR37","first-page":"2051","volume":"20","author":"Y Shekofteh","year":"2013","unstructured":"Shekofteh, Y., & Almasganj, F. (2013). Remote diagnosis of unilateral vocal fold paralysis using matching pursuit based features extracted from telephony speech signal. Scientia Iranica, 20(6), 2051\u20132060.","journal-title":"Scientia Iranica"},{"key":"10134_CR38","doi-asserted-by":"publisher","first-page":"49667","DOI":"10.1109\/ACCESS.2024.3371713","volume":"12","author":"I Sindhu","year":"2024","unstructured":"Sindhu, I., & Sainin, M. S. (2024). Automatic speech and voice disorder detection using deep learning\u2014a systematic literature review. IEEE Access, 12, 49667\u201349681.","journal-title":"IEEE Access"},{"issue":"1","key":"10134_CR39","first-page":"1929","volume":"15","author":"N Srivastava","year":"2014","unstructured":"Srivastava, N., Hinton, G., Krizhevsky, A., Sutskever, I., & Salakhutdinov, R. (2014). Dropout: A simple way to prevent neural networks from overfitting. The Journal of Machine Learning Research, 15(1), 1929\u20131958.","journal-title":"The Journal of Machine Learning Research"},{"key":"10134_CR40","doi-asserted-by":"publisher","first-page":"1","DOI":"10.1155\/2021\/6635964","volume":"2021","author":"SA Syed","year":"2021","unstructured":"Syed, S. A., Rashid, M., Hussain, S., & Zahid, H. (2021). Comparative analysis of CNN and RNN for voice pathology detection. BioMed Research International, 2021, 1\u20138.","journal-title":"BioMed Research International"},{"issue":"1","key":"10134_CR41","doi-asserted-by":"publisher","first-page":"22719","DOI":"10.1038\/s41598-023-49869-6","volume":"13","author":"V Verma","year":"2023","unstructured":"Verma, V., Benjwal, A., Chhabra, A., Singh, S. K., Kumar, S., Gupta, B. B., Arya, V., & Chui, K. T. (2023). A novel hybrid model integrating MFCC and acoustic parameters for voice disorder detection. Scientific Reports, 13(1), 22719.","journal-title":"Scientific Reports"},{"key":"10134_CR42","doi-asserted-by":"publisher","first-page":"7814952","DOI":"10.1155\/2022\/7814952","volume":"2022","author":"M Zakariah","year":"2022","unstructured":"Zakariah, M., Ajmi Alotaibi, Y., Guo, Y., Tran-Trung, K., & Elahi, M. M. (2022). An analytical study of speech pathology detection based on MFCC and deep neural networks. Computational and Mathematical Methods in Medicine, 2022, 7814952.","journal-title":"Computational and Mathematical Methods in Medicine"},{"key":"10134_CR43","volume-title":"Dive into deep learning","author":"A Zhang","year":"2023","unstructured":"Zhang, A., Lipton, Z. C., Li, M., & Smola, A. J. (2023). Dive into deep learning. Cambridge University Press."},{"key":"10134_CR44","doi-asserted-by":"publisher","first-page":"105624","DOI":"10.1016\/j.bspc.2023.105624","volume":"88","author":"D Zhao","year":"2024","unstructured":"Zhao, D., Qiu, Z., Jiang, Y., Zhu, X., Zhang, X., & Tao, Z. (2024). A depthwise separable CNN-based interpretable feature extraction network for automatic pathological voice detection. Biomedical Signal Processing and Control, 88, 105624.","journal-title":"Biomedical Signal Processing and Control"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10134-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10134-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10134-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,9,12]],"date-time":"2024-09-12T12:12:29Z","timestamp":1726143149000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10134-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,8,10]]},"references-count":44,"journal-issue":{"issue":"3","published-print":{"date-parts":[[2024,9]]}},"alternative-id":["10134"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10134-4","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,8,10]]},"assertion":[{"value":"3 April 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"28 July 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 August 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"The authors have no competing interests to declare that are relevant to the content of this paper.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}}]}}