{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,4,1]],"date-time":"2026-04-01T17:26:10Z","timestamp":1775064370470,"version":"3.50.1"},"reference-count":35,"publisher":"Springer Science and Business Media LLC","issue":"7","license":[{"start":{"date-parts":[[2025,3,11]],"date-time":"2025-03-11T00:00:00Z","timestamp":1741651200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2025,3,11]],"date-time":"2025-03-11T00:00:00Z","timestamp":1741651200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["Circuits Syst Signal Process"],"published-print":{"date-parts":[[2025,7]]},"DOI":"10.1007\/s00034-025-03037-5","type":"journal-article","created":{"date-parts":[[2025,3,11]],"date-time":"2025-03-11T10:26:20Z","timestamp":1741688780000},"page":"5044-5074","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":1,"title":["Optimized Deep Embedded Clustering-Based Speaker Diarization with Speech Enhancement"],"prefix":"10.1007","volume":"44","author":[{"given":"S. Merlin","family":"Revathy","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"S. S.","family":"Kumar","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2025,3,11]]},"reference":[{"key":"3037_CR1","doi-asserted-by":"publisher","first-page":"126671","DOI":"10.1109\/ACCESS.2020.3007312","volume":"8","author":"R Ahmad","year":"2020","unstructured":"R. Ahmad, S. Zubair, H. Alquhayz, Speech enhancement for multimodal speaker diarization system. IEEE Access 8, 126671\u2013126680 (2020)","journal-title":"IEEE Access"},{"key":"3037_CR2","doi-asserted-by":"publisher","first-page":"5163","DOI":"10.3390\/s19235163","volume":"19","author":"R Ahmad","year":"2019","unstructured":"R. Ahmad, S. Zubair, H. Alquhayz, A. Ditta, Multimodal speaker diarization using a pre-trained audio-visual synchronization model. Sensors (Basel) 19, 5163 (2019)","journal-title":"Sensors (Basel)"},{"key":"3037_CR3","doi-asserted-by":"publisher","first-page":"356","DOI":"10.1109\/TASL.2011.2125954","volume":"20","author":"X Anguera","year":"2012","unstructured":"X. Anguera, S. Bozonnet, N. Evans, C. Fredouille, G. Friedland, O. Vinyals, Speaker diarization: a review of recent research. IEEE Trans. Audio Speech Lang. Process. 20, 356\u2013370 (2012)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"3037_CR4","doi-asserted-by":"publisher","first-page":"105709","DOI":"10.1016\/j.knosys.2020.105709","volume":"195","author":"Q Askari","year":"2020","unstructured":"Q. Askari, I. Younas, M. Saeed, Political optimizer: a novel socio-inspired meta-heuristic for global optimization. Knowl.-Based Syst. 195, 105709 (2020)","journal-title":"Knowl.-Based Syst."},{"key":"3037_CR5","doi-asserted-by":"publisher","unstructured":"P. Assmann, Q. Summerfield. The Perception of Speech Under Adverse Conditions. in Speech Processing in the Auditory System, 231\u2013308 (Springer, New York, NY, 2004). https:\/\/doi.org\/10.1007\/0-387-21575-1_5.","DOI":"10.1007\/0-387-21575-1_5"},{"key":"3037_CR6","doi-asserted-by":"publisher","unstructured":"P. Cyrta, T. Trzci\u0144ski, W. Stokowiec. Speaker Diarization Using Deep Recurrent Convolutional Neural Networks for Speaker Embeddings. in Information Systems Architecture and Technology: Proceedings of 38th International Conference on Information Systems Architecture and Technology \u2013 ISAT 2017 107\u2013117 (Springer International Publishing, Cham, 2018). https:\/\/doi.org\/10.1007\/978-3-319-67220-5_10","DOI":"10.1007\/978-3-319-67220-5_10"},{"key":"3037_CR7","doi-asserted-by":"crossref","unstructured":"Y. Du, W. Hu, Y. Yan, T. Wang, Y. Zhang. Audio Segmentation via Tri-Model Bayesian Information Criterion. in 2007 IEEE International Conference on Acoustics, Speech and Signal Processing-ICASSP'07.- vol. 1 I-205-I\u2013208 (2007)","DOI":"10.1109\/ICASSP.2007.366652"},{"key":"3037_CR8","unstructured":"L. Feng , L. K. Hansen. A New Database for Speaker Recognition. A New Database for Speaker Recognition, 3662 (2005)"},{"key":"3037_CR9","doi-asserted-by":"publisher","unstructured":"Y. Fujita, N. Kanda, S. Horiguchi, K. Nagamatsu, S. Watanabe. End-to-End Neural Speaker Diarization with Permutation-Free Objectives. (2019) https:\/\/doi.org\/10.48550\/arXiv.1909.05952.","DOI":"10.48550\/arXiv.1909.05952"},{"key":"3037_CR10","doi-asserted-by":"publisher","first-page":"22","DOI":"10.1121\/1.383075","volume":"66","author":"SA Gelfand","year":"1979","unstructured":"S.A. Gelfand, S. Silman, Effects of small room reverberation upon the recognition of some consonant features. J. Acoust. Soc. Am. 66, 22\u201329 (1979)","journal-title":"J. Acoust. Soc. Am."},{"key":"3037_CR11","doi-asserted-by":"publisher","first-page":"139","DOI":"10.1109\/91.917120","volume":"9","author":"C-F Juang","year":"2001","unstructured":"C.-F. Juang, C.-T. Lin, Noisy speech processing by recurrently adaptive fuzzy filters. IEEE Trans. Fuzzy Syst. 9, 139\u2013152 (2001)","journal-title":"IEEE Trans. Fuzzy Syst."},{"key":"3037_CR12","doi-asserted-by":"publisher","unstructured":"D. Klement, M. Diez, F. Landini, L. Burget, A. Silnova, M. Delcroix, N. Tawara. Discriminative Training of VBx Diarization. in ICASSP 2024-2024 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP) 11871\u201311875 (2024). https:\/\/doi.org\/10.1109\/ICASSP48485.2024.10446119","DOI":"10.1109\/ICASSP48485.2024.10446119"},{"key":"3037_CR13","doi-asserted-by":"publisher","first-page":"101254","DOI":"10.1016\/j.csl.2021.101254","volume":"71","author":"F Landini","year":"2022","unstructured":"F. Landini, J. Profant, M. Diez, L. Burget, Bayesian HMM clustering of x-vector sequences (VBx) in speaker diarization: theory, implementation and analysis on standard tasks. Comput. Speech Lang. 71, 101254 (2022)","journal-title":"Comput. Speech Lang."},{"key":"3037_CR14","first-page":"111","volume":"38","author":"H Levitt","year":"2001","unstructured":"H. Levitt, Noise reduction in hearing aids: a review. J. Rehabil. Res. Dev. 38, 111\u2013121 (2001)","journal-title":"J. Rehabil. Res. Dev."},{"key":"3037_CR15","doi-asserted-by":"crossref","unstructured":"D. Michelsanti, Z.-H. Tan, S.-X. Zhang, Y. Xu, M. Yu, D. Yu, J. Jensen. An Overview of Deep-Learning-Based Audio-Visual Speech Enhancement and Separation. IEEE\/ACM Transactions on Audio, Speech, and Language Processing 29, 1368\u20131396 (2021)","DOI":"10.1109\/TASLP.2021.3066303"},{"key":"3037_CR16","doi-asserted-by":"publisher","first-page":"476","DOI":"10.1121\/1.396880","volume":"84","author":"AK Nabelek","year":"1988","unstructured":"A.K. Nabelek, Identification of vowels in quiet, noise, and reverberation: relationships with age and hearing loss. J. Acoust. Soc. Am. 84, 476\u2013484 (1988)","journal-title":"J. Acoust. Soc. Am."},{"key":"3037_CR17","doi-asserted-by":"publisher","first-page":"1204","DOI":"10.1109\/TASLP.2021.3061885","volume":"29","author":"M Pal","year":"2021","unstructured":"M. Pal, M. Kumar, R. Peri, T.J. Park, S.H. Kim, C. Lord, S. Bishop, S. Narayanan, Meta-learning with latent space clustering in generative adversarial network for speaker diarization. IEEE Trans. Audio Speech Lang. Process. 29, 1204\u20131219 (2021)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"3037_CR18","doi-asserted-by":"publisher","DOI":"10.48550\/arXiv.2101.09624","author":"TJ Park","year":"2021","unstructured":"T.J. Park, N. Kanda, D. Dimitriadis, K.J. Han, S. Watanabe, S. Narayanan, A review of speaker diarization: recent advances with deep learning. Comput. Speech Lang. (2021). https:\/\/doi.org\/10.48550\/arXiv.2101.09624","journal-title":"Comput. Speech Lang."},{"key":"3037_CR19","doi-asserted-by":"publisher","unstructured":"S. Pascual, A. Bonafonte, J. Serra. SEGAN: Speech Enhancement Generative Adversarial Network. in 3642\u20133646 (2017). https:\/\/doi.org\/10.21437\/Interspeech.2017-1428.","DOI":"10.21437\/Interspeech.2017-1428"},{"key":"3037_CR20","doi-asserted-by":"publisher","unstructured":"G. Pechetti, A. R. D. Bhavani, A. Dayal, S. Ponnada. Unraveling the Techniques for Speaker Diarization. in Cognitive Computing and Cyber Physical Systems, 300\u2013308 (Springer Nature Switzerland, Cham, 2024). https:\/\/doi.org\/10.1007\/978-3-031-48888-7_25.","DOI":"10.1007\/978-3-031-48888-7_25"},{"key":"3037_CR21","doi-asserted-by":"publisher","first-page":"47","DOI":"10.5815\/ijigsp.2013.05.06","volume":"5","author":"R Ramani","year":"2013","unstructured":"R. Ramani, N.S. Vanitha, S. Valarmathy, The pre-processing techniques for breast cancer detection in mammography images. Inter. J. Image Graphics Sig. Process. 5, 47 (2013)","journal-title":"Inter. J. Image Graphics Sig. Process."},{"key":"3037_CR22","doi-asserted-by":"publisher","first-page":"631","DOI":"10.1007\/s12065-020-00378-9","volume":"13","author":"V Sethuram","year":"2020","unstructured":"V. Sethuram, A. Prasad, R.R. Rao, Optimal trained artificial neural network for Telugu speaker diarization. Evol. Intel. 13, 631\u2013648 (2020)","journal-title":"Evol. Intel."},{"key":"3037_CR23","doi-asserted-by":"crossref","unstructured":"P. Singh , S. Ganapathy. Self-Supervised Representation Learning With Path Integral Clustering for Speaker Diarization. in IEEE\/ACM Transactions on Audio, Speech, and Language Processing. 29, 1639\u20131649 (2021).","DOI":"10.1109\/TASLP.2021.3075100"},{"key":"3037_CR24","doi-asserted-by":"publisher","first-page":"249","DOI":"10.1016\/S0167-6393(98)00019-3","volume":"24","author":"IY Soon","year":"1998","unstructured":"I.Y. Soon, S.N. Koh, C.K. Yeo, Noisy speech enhancement using discrete cosine transform1. Speech Commun. 24, 249\u2013257 (1998)","journal-title":"Speech Commun."},{"key":"3037_CR25","doi-asserted-by":"publisher","first-page":"97","DOI":"10.1007\/s00034-013-9633-0","volume":"33","author":"V Stojanovic","year":"2014","unstructured":"V. Stojanovic, V. Filipovic, Adaptive input design for identification of output error model with constrained output. Circuits Syst. Signal Process 33, 97\u2013113 (2014)","journal-title":"Circuits Syst. Signal Process"},{"key":"3037_CR26","doi-asserted-by":"publisher","first-page":"3974","DOI":"10.1002\/rnc.3544","volume":"26","author":"V Stojanovic","year":"2016","unstructured":"V. Stojanovic, N. Nedic, Identification of time-varying OE models in presence of non-Gaussian noise: application to pneumatic servo drives. Inter. J. Robust Nonlinear Control 26, 3974\u20133995 (2016)","journal-title":"Inter. J. Robust Nonlinear Control"},{"key":"3037_CR27","doi-asserted-by":"publisher","first-page":"255","DOI":"10.1016\/j.aej.2016.12.009","volume":"57","author":"V Subba Ramaiah","year":"2018","unstructured":"V. Subba Ramaiah, R. Rajeswara Rao, Speaker diarization system using HXLPS and deep neural network. Alex. Eng. J. 57, 255\u2013266 (2018)","journal-title":"Alex. Eng. J."},{"key":"3037_CR28","doi-asserted-by":"publisher","first-page":"1309","DOI":"10.1121\/1.401923","volume":"90","author":"HM Sussman","year":"1991","unstructured":"H.M. Sussman, H.A. McCaffrey, S.A. Matthews, An investigation of locus equations as a source of relational invariance for stop place categorization. J. Acoust. Soc. Am. 90, 1309\u20131325 (1991)","journal-title":"J. Acoust. Soc. Am."},{"key":"3037_CR29","doi-asserted-by":"crossref","unstructured":"K. Tan, D. Wang. Towards Model Compression for Deep Learning Based Speech Enhancement. in IEEE\/ACM transactions on audio, speech, and language processing 29, 1785\u20131794 (2021)","DOI":"10.1109\/TASLP.2021.3082282"},{"key":"3037_CR30","doi-asserted-by":"publisher","first-page":"1557","DOI":"10.1109\/TASL.2006.878256","volume":"14","author":"SE Tranter","year":"2006","unstructured":"S.E. Tranter, D.A. Reynolds, An overview of automatic speaker diarization systems. IEEE Trans. Audio Speech Lang. Process. 14, 1557\u20131565 (2006)","journal-title":"IEEE Trans. Audio Speech Lang. Process."},{"key":"3037_CR31","doi-asserted-by":"publisher","unstructured":"C. Veaux, J. Yamagishi, K. MacDonald. SUPERSEDED - CSTR VCTK Corpus: English Multi-speaker Corpus for CSTR Voice Cloning Toolkit. (2017) https:\/\/doi.org\/10.7488\/ds\/1994","DOI":"10.7488\/ds\/1994"},{"key":"3037_CR32","unstructured":"J. Xie, R. Girshick, A. Farhadi. Unsupervised deep embedding for clustering analysis. in Proceedings of the 33rd International conference on machine learning - Volume 48 478\u2013487 (JMLR.org, New York, NY, USA, 2016)"},{"key":"3037_CR33","doi-asserted-by":"publisher","unstructured":"C. Yang,Y. Wang. Robust End-to-end Speaker Diarization with Generic Neural Clustering. in 1471\u20131475 (2022). https:\/\/doi.org\/10.21437\/Interspeech.2022-10404.","DOI":"10.21437\/Interspeech.2022-10404"},{"key":"3037_CR34","doi-asserted-by":"publisher","first-page":"2188","DOI":"10.1109\/TASLP.2017.2747097","volume":"25","author":"C Yu","year":"2017","unstructured":"C. Yu, J.H.L. Hansen, Active learning based constrained clustering for speaker diarization. IEEE Trans. Audio, Speech, Lang. Process. 25, 2188\u20132198 (2017)","journal-title":"IEEE Trans. Audio, Speech, Lang. Process."},{"key":"3037_CR35","doi-asserted-by":"publisher","first-page":"47963","DOI":"10.1109\/ACCESS.2020.2978435","volume":"8","author":"Z Zhu","year":"2020","unstructured":"Z. Zhu, Y. Sun, J. White, Z. Chang, S. Pang, Signal retrieval with measurement system knowledge using variational generative model. IEEE Access 8, 47963\u201347972 (2020)","journal-title":"IEEE Access"}],"container-title":["Circuits, Systems, and Signal Processing"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-03037-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s00034-025-03037-5\/fulltext.html","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s00034-025-03037-5.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2025,7,1]],"date-time":"2025-07-01T21:01:43Z","timestamp":1751403703000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s00034-025-03037-5"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2025,3,11]]},"references-count":35,"journal-issue":{"issue":"7","published-print":{"date-parts":[[2025,7]]}},"alternative-id":["3037"],"URL":"https:\/\/doi.org\/10.1007\/s00034-025-03037-5","relation":{},"ISSN":["0278-081X","1531-5878"],"issn-type":[{"value":"0278-081X","type":"print"},{"value":"1531-5878","type":"electronic"}],"subject":[],"published":{"date-parts":[[2025,3,11]]},"assertion":[{"value":"31 May 2024","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 February 2025","order":2,"name":"revised","label":"Revised","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"7 February 2025","order":3,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"11 March 2025","order":4,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}}]}}