{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,8,1]],"date-time":"2026-08-01T09:48:15Z","timestamp":1785577695020,"version":"3.56.0"},"reference-count":23,"publisher":"Springer Science and Business Media LLC","issue":"4","license":[{"start":{"date-parts":[[2024,10,10]],"date-time":"2024-10-10T00:00:00Z","timestamp":1728518400000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2024,10,10]],"date-time":"2024-10-10T00:00:00Z","timestamp":1728518400000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Int J Speech Technol"],"published-print":{"date-parts":[[2024,12]]},"DOI":"10.1007\/s10772-024-10150-4","type":"journal-article","created":{"date-parts":[[2024,10,10]],"date-time":"2024-10-10T14:03:15Z","timestamp":1728568995000},"page":"1013-1026","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":13,"title":["Enhanced text-independent speaker recognition using MFCC, Bi-LSTM, and CNN-based noise removal techniques"],"prefix":"10.1007","volume":"27","author":[{"given":"Manish","family":"Tiwari","sequence":"first","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Deepak Kumar","family":"Verma","sequence":"additional","affiliation":[],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"297","published-online":{"date-parts":[[2024,10,10]]},"reference":[{"issue":"21","key":"10150_CR1","doi-asserted-by":"publisher","first-page":"33111","DOI":"10.1007\/s11042-023-14942-9","volume":"82","author":"R Chakroun","year":"2023","unstructured":"Chakroun, R., & Frikha, M. (2023). A deep learning approach for text-independent speaker recognition with short utterances. Multimedia Tools and Applications, 82(21), 33111\u201333133.","journal-title":"Multimedia Tools and Applications"},{"issue":"7","key":"10150_CR2","doi-asserted-by":"publisher","first-page":"3461","DOI":"10.3390\/s23073461","volume":"23","author":"G Costantini","year":"2023","unstructured":"Costantini, G., Cesarini, V., & Brenna, E. (2023). High-level CNN and machine learning methods for speaker recognition. Sensors, 23(7), 3461.","journal-title":"Sensors"},{"key":"10150_CR3","first-page":"1","volume":"1","author":"A Das","year":"2014","unstructured":"Das, A., Jena, M. R., & Barik, K. K. (2014). Mel-frequency cepstral coefficient (MFCC) a novel method for speaker recognition. Digital Technologies, 1, 1\u20133.","journal-title":"Digital Technologies"},{"key":"10150_CR4","doi-asserted-by":"publisher","first-page":"24013","DOI":"10.1007\/s11042-019-08293-7","volume":"79","author":"SA El-Moneim","year":"2020","unstructured":"El-Moneim, S. A., Nassar, M. A., Dessouky, M. I., Ismail, N. A., El-Fishawy, A. S., & Abd El-Samie, F. E. (2020). Text-independent speaker recognition using LSTM-RNN and speech enhancement. Multimedia Tools and Applications, 79, 24013\u201324028.","journal-title":"Multimedia Tools and Applications"},{"key":"10150_CR5","doi-asserted-by":"crossref","unstructured":"Fang, H., & Gerkmann, T. (2023) Uncertainty estimation in deep speech enhancement using complex Gaussian mixture models. In 2023 IEEE international conference on acoustics, speech and signal processing (ICASSP 2023) (pp. 1\u20135. IEEE).","DOI":"10.1109\/ICASSP49357.2023.10095213"},{"key":"10150_CR6","doi-asserted-by":"publisher","DOI":"10.1016\/j.jisa.2023.103665","volume":"80","author":"P Gambhir","year":"2024","unstructured":"Gambhir, P., Dev, A., Bansal, P., Sharma, D. K., & Gupta, D. (2024). Residual networks for text-independent speaker identification: Unleashing the power of residual learning. Journal of Information Security and Applications, 80, 103665.","journal-title":"Journal of Information Security and Applications"},{"issue":"17","key":"10150_CR7","doi-asserted-by":"publisher","first-page":"9787","DOI":"10.3390\/app13179787","volume":"13","author":"X Guo","year":"2023","unstructured":"Guo, X., Qin, X., Zhang, Q., Zhang, Y., Wang, P., & Fan, Z. (2023). Speaker recognition based on dung beetle optimized CNN. Applied Sciences, 13(17), 9787.","journal-title":"Applied Sciences"},{"key":"10150_CR8","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-023-17017-x","author":"A Gupta","year":"2023","unstructured":"Gupta, A., & Archana, P. (2023). Speech refinement using Bi-LSTM and improved spectral clustering in speaker diarization. Multimedia Tools and Applications. https:\/\/doi.org\/10.1007\/s11042-023-17017-x","journal-title":"Multimedia Tools and Applications"},{"issue":"1","key":"10150_CR9","doi-asserted-by":"publisher","first-page":"123","DOI":"10.1007\/s10772-019-09665-y","volume":"23","author":"S Hourri","year":"2020","unstructured":"Hourri, S., & Jamal, K. (2020). A deep learning approach for speaker recognition. International Journal of Speech Technology, 23(1), 123\u2013131.","journal-title":"International Journal of Speech Technology"},{"key":"10150_CR10","unstructured":"http:\/\/www.imagicdatatech.com\/index.php\/home\/dataopensource\/data_info\/id\/101,"},{"key":"10150_CR11","unstructured":"Kim, H. S. (2023). Linear predictive coding is all-pole resonance modeling.\u00a0Center for Computer Research in Music and Acoustics, Stanford University."},{"key":"10150_CR12","doi-asserted-by":"crossref","unstructured":"Li, K.P., Wrench, K.H. (1983). An approach to text-independent speaker recognition with short utterances. In IEEE international conference on acoustics, speech, and signal processing (ICASSP 1983) (pp. 555\u2013558). IEEE.","DOI":"10.1109\/ICASSP.1983.1172258"},{"key":"10150_CR13","first-page":"50221","volume":"36","author":"T Liu","year":"2023","unstructured":"Liu, T., Lee, K. A., Wang, Q., & Li, H. (2023). Disentangling voice and content with self-supervision for speaker recognition. Advances in Neural Information Processing Systems, 36, 50221\u201350236.","journal-title":"Advances in Neural Information Processing Systems"},{"key":"10150_CR14","doi-asserted-by":"crossref","unstructured":"Nilufar, S., Ray, N., Islam Molla, M.K., & Hirose, K. (2012). Spectrogram based features selection using multiple kernel learning for speech\/music discrimination. In IEEE international conference on acoustics, speech and signal processing (ICASSP) (pp 501\u2013504).","DOI":"10.1109\/ICASSP.2012.6287926"},{"key":"10150_CR15","doi-asserted-by":"crossref","unstructured":"Parada, P. P., Sharma, D., Naylor, P. A., & van Waterschoot, T. (2014). Reverberant-speech-recognition:-A-phoneme-analysis. In Proceedings of IEEE global conference on signal and information processing, (GlobalSIP) (pp 567\u2013571).","DOI":"10.1109\/GlobalSIP.2014.7032181"},{"key":"10150_CR16","doi-asserted-by":"publisher","DOI":"10.1016\/j.eswa.2023.120608","volume":"230","author":"NT Pham","year":"2023","unstructured":"Pham, N. T., Dang, D. N. M., Nguyen, N. D., Nguyen, T. T., Nguyen, H., Manavalan, B., & Nguyen, S. D. (2023). Hybrid data augmentation and deep attention-based dilated convolutional-recurrent neural networks for speech emotion recognition. Expert Systems with Applications, 230, 120608.","journal-title":"Expert Systems with Applications"},{"key":"10150_CR17","doi-asserted-by":"crossref","unstructured":"Safriadi, S., Mahlil, M., Hidayat, H. T., Nasir, M., & Anwar, A. (2023). The classification of emotion based on human voice by using Mel Frequency Cepstrum Coefficient (MFCC) and Naive Bayes method. In AIP conference proceedings, (Vol. 2431, no. 1). AIP Publishing.","DOI":"10.1063\/5.0117958"},{"key":"10150_CR18","doi-asserted-by":"publisher","first-page":"164","DOI":"10.3390\/electronics8020164","volume":"8","author":"Y Seo","year":"2019","unstructured":"Seo, Y., & Huh, J. (2019). Automatic emotion-based music classification for supporting intelligent IoT applications. Electronics, 8, 164.","journal-title":"Electronics"},{"issue":"1","key":"10150_CR19","first-page":"3111","volume":"14","author":"M Shafieian","year":"2023","unstructured":"Shafieian, M. (2023). Hidden Markov model and Persian speech recognition. International Journal of Nonlinear Analysis and Applications, 14(1), 3111\u20133119.","journal-title":"International Journal of Nonlinear Analysis and Applications"},{"key":"10150_CR20","doi-asserted-by":"publisher","DOI":"10.1007\/s11042-023-17573-2","author":"MK Singh","year":"2023","unstructured":"Singh, M. K. (2023). A text independent speaker identification system using ANN, RNN, and CNN classification technique. Multimedia Tools and Applications. https:\/\/doi.org\/10.1007\/s11042-023-17573-2","journal-title":"Multimedia Tools and Applications"},{"key":"10150_CR21","doi-asserted-by":"publisher","first-page":"5328","DOI":"10.1109\/ACCESS.2023.3236242","volume":"11","author":"R Soleymanpour","year":"2023","unstructured":"Soleymanpour, R., Soleymanpour, M., Brammer, A. J., Johnson, M. T., & Kim, I. (2023). Speech enhancement algorithm based on a convolutional neural network reconstruction of the temporal envelope of speech in noisy environments. IEEE Access, 11, 5328\u20135336.","journal-title":"IEEE Access"},{"key":"10150_CR22","doi-asserted-by":"publisher","first-page":"23","DOI":"10.1109\/MCAS.2011.941079","volume":"11","author":"R Togneri","year":"2011","unstructured":"Togneri, R., & Pullella, D. (2011). An overview of speaker identification: Accuracy and robustness issues. IEEE Circuits and Systems Magazine, 11, 23\u201361.","journal-title":"IEEE Circuits and Systems Magazine"},{"key":"10150_CR23","doi-asserted-by":"crossref","unstructured":"Vaessen, N., & Van Leeuwen, D. A. (2022) Fine-tuning wav2vec2 for speaker recognition. In 2022 IEEE international conference on acoustics, speech and signal processing (ICASSP 2022) (pp. 7967\u20137971). IEEE.","DOI":"10.1109\/ICASSP43922.2022.9746952"}],"container-title":["International Journal of Speech Technology"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10150-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s10772-024-10150-4\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s10772-024-10150-4.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2024,12,16]],"date-time":"2024-12-16T09:59:40Z","timestamp":1734343180000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s10772-024-10150-4"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2024,10,10]]},"references-count":23,"journal-issue":{"issue":"4","published-print":{"date-parts":[[2024,12]]}},"alternative-id":["10150"],"URL":"https:\/\/doi.org\/10.1007\/s10772-024-10150-4","relation":{},"ISSN":["1381-2416","1572-8110"],"issn-type":[{"value":"1381-2416","type":"print"},{"value":"1572-8110","type":"electronic"}],"subject":[],"published":{"date-parts":[[2024,10,10]]},"assertion":[{"value":"10 July 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 September 2024","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"10 October 2024","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"Authors of the paper declare no conflict of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of interest"}}]}}