{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T18:30:08Z","timestamp":1772821808773,"version":"3.50.1"},"reference-count":20,"publisher":"Springer Science and Business Media LLC","issue":"3","license":[{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T00:00:00Z","timestamp":1772755200000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":["SN COMPUT. SCI."],"DOI":"10.1007\/s42979-026-04819-7","type":"journal-article","created":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T12:36:51Z","timestamp":1772800611000},"update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["Integrating Large Language Models for Enhanced Speaker Diarization in Overlapping Speech Scenarios"],"prefix":"10.1007","volume":"7","author":[{"given":"Devi Venkata Revathi","family":"Poduri","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Swarna","family":"Kuchibhotla","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Vamsi Aditya","family":"Tummala","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Sagar Imambi","family":"Shaik","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"given":"Hima Deepthi","family":"Vankayalapati","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,3,6]]},"reference":[{"key":"4819_CR1","unstructured":"Wooters C, Huijbregts M. The ICSI RT07s speaker diarization system. In: International Evaluation Workshop on Rich Transcription. Berlin, Heidelberg: Springer Berlin Heidelberg; 2007."},{"issue":"4","key":"4819_CR2","doi-asserted-by":"publisher","DOI":"10.3390\/app15042002","volume":"15","author":"D O\u2019Shaughnessy","year":"2025","unstructured":"O\u2019Shaughnessy D. Speaker diarization: a review of objectives and methods. Appl Sci. 2025;15(4):2002.","journal-title":"Appl Sci"},{"key":"4819_CR3","doi-asserted-by":"crossref","unstructured":"Xiao X, Chen Z, Zhou T, Gong Y. Microsoft speaker diarization system for the VoxCeleb challenge 2020. In: Proceedings of the IEEE International Conference on acoustics, speech and signal processing (ICASSP), Toronto, Canada, 2021; pp. 5824\u20135828.","DOI":"10.1109\/ICASSP39728.2021.9413832"},{"key":"4819_CR4","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2021.101317","volume":"72","author":"TJ Park","year":"2022","unstructured":"Park TJ, Dimitriadis D, Han KJ, Watanabe S, Narayanan S. A review of speaker diarization: recent advances with deep learning. Comput Speech Lang. 2022;72:101317.","journal-title":"Comput Speech Lang"},{"key":"4819_CR5","doi-asserted-by":"crossref","unstructured":"Coria JM, et al. Overlap-aware low-latency online speaker diarization based on end-to-end local segmentation. In: 2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU). IEEE, 2021.","DOI":"10.1109\/ASRU51503.2021.9688044"},{"key":"4819_CR6","doi-asserted-by":"crossref","unstructured":"Ayasi A, Joshy J, Rajan R. Speaker diarization using BiLSTM and BiGRU with self-attention. In: 2022 Second International Conference on Next Generation Intelligent Systems (ICNGIS), Kottayam, India, 2022; pp. 1\u20135.","DOI":"10.1109\/ICNGIS54955.2022.10079831"},{"key":"4819_CR7","doi-asserted-by":"crossref","unstructured":"Desplanques B, Thienpondt J, Demuynck K. ECAPA-TDNN: Emphasized channel attention, propagation and aggregation in TDNN. In: Proceedings of Interspeech, 2020; pp. 3830\u20133834.","DOI":"10.21437\/Interspeech.2020-2650"},{"key":"4819_CR8","first-page":"3128","volume":"29","author":"S Hahm","year":"2021","unstructured":"Hahm S. Transformer-based speaker embedding for speaker diarization. IEEE\/ACM Trans Audio Speech Lang Process. 2021;29:3128\u201339.","journal-title":"IEEE\/ACM Trans Audio Speech Lang Process"},{"key":"4819_CR9","doi-asserted-by":"crossref","unstructured":"Dawalatabad N, Ravanelli M, Grondin F, Na H. ECAPA-TDNN embeddings for speaker diarization. 2021. arXiv preprint arXiv:2104.01466.","DOI":"10.21437\/Interspeech.2021-941"},{"key":"4819_CR10","unstructured":"Aloradi A, Elminshawi M, Chetupalli SR, Habets EAP. Target-speaker voice activity detection in multi-talker scenarios: an empirical study,\u00a0Speech Communication. In: 15th ITG Conference, Aachen, 2023; pp. 250\u2013254."},{"key":"4819_CR11","unstructured":"Wang Q, et al. Speaker diarization with LSTM neural networks. In: Proceedings of the IEEE Automatic Speech Recognition and Understanding Workshop, IEEE, 2017; pp. 1\u20136."},{"key":"4819_CR12","volume":"67","author":"J Bullock","year":"2021","unstructured":"Bullock J, Kautz H. Meeting diarization in conversational speech. Comput Speech Lang. 2021;67:101173.","journal-title":"Comput Speech Lang"},{"issue":"1","key":"4819_CR13","doi-asserted-by":"publisher","DOI":"10.1038\/s41598-025-09385-1","volume":"15","author":"M Ahmed","year":"2025","unstructured":"Ahmed M, et al. An enhanced deep learning approach for speaker diarization using TitaNet, MarbelNet and time delay network. Sci Rep. 2025;15(1):24501.","journal-title":"Sci Rep"},{"key":"4819_CR14","unstructured":"Zuluaga-Gomez J, Vesel\u00fd K, Sz\u00f6ke I, Klakow D. ATCO2 corpus: a large-scale dataset for ASR and natural language understanding of air traffic control communications. 2022. arXiv preprint arXiv:2211.04054."},{"key":"4819_CR15","unstructured":"Wang J, Dudy Sh, He X, Wang Z, Southwell R, Whitehill J. Speaker diarization in the classroom: how much does each student speak in group discussions? In: 17th International Educational Data Mining Conference (EDM 2024), 360\u2014367 (2024)"},{"issue":"24","key":"4819_CR16","doi-asserted-by":"publisher","DOI":"10.3390\/math13243950","volume":"13","author":"C Sun","year":"2025","unstructured":"Sun C, Sun M, Wei W. Improving speaker diarization for overlapped speech with texture-aware feature fusion. Mathematics. 2025;13(24):3950.","journal-title":"Mathematics"},{"key":"4819_CR17","unstructured":"Vaswani A, Shazeer N, Parmar N, Uszkoreit J. Attention is all you need. In: Advances in neural information processing systems (NeurIPS), 2017; pp. 5998\u20136008."},{"key":"4819_CR18","doi-asserted-by":"crossref","first-page":"1485","DOI":"10.1109\/LSP.2020.3016837","volume":"27","author":"D Garcia-Romero","year":"2020","unstructured":"Garcia-Romero D, McCree A. Speaker diarization using neural embeddings. IEEE Signal Process Lett. 2020;27:1485\u20139.","journal-title":"IEEE Signal Process Lett"},{"key":"4819_CR19","first-page":"36","volume":"124","author":"C Xu","year":"2020","unstructured":"Xu C, Wang S, Xu B. Speaker diarization with contextual neural representations. Speech Commun. 2020;124:36\u201345.","journal-title":"Speech Commun"},{"issue":"4","key":"4819_CR20","first-page":"84","volume":"37","author":"J Li","year":"2020","unstructured":"Li J, Deng L, Gong Y. Advances in deep learning for speaker recognition and diarization. IEEE Signal Process Mag. 2020;37(4):84\u201395.","journal-title":"IEEE Signal Process Mag"}],"container-title":["SN Computer Science"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-026-04819-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/article\/10.1007\/s42979-026-04819-7","content-type":"text\/html","content-version":"vor","intended-application":"text-mining"},{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/s42979-026-04819-7.pdf","content-type":"application\/pdf","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,3,6]],"date-time":"2026-03-06T12:36:55Z","timestamp":1772800615000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/s42979-026-04819-7"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026,3,6]]},"references-count":20,"journal-issue":{"issue":"3","published-online":{"date-parts":[[2026,3]]}},"alternative-id":["4819"],"URL":"https:\/\/doi.org\/10.1007\/s42979-026-04819-7","relation":{},"ISSN":["2661-8907"],"issn-type":[{"value":"2661-8907","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026,3,6]]},"assertion":[{"value":"26 August 2025","order":1,"name":"received","label":"Received","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"9 February 2026","order":2,"name":"accepted","label":"Accepted","group":{"name":"ArticleHistory","label":"Article History"}},{"value":"6 March 2026","order":3,"name":"first_online","label":"First Online","group":{"name":"ArticleHistory","label":"Article History"}},{"order":1,"name":"Ethics","group":{"name":"EthicsHeading","label":"Declarations"}},{"value":"As the corresponding author of this manuscript, the authors and Co-authors declare that they have no conflicts of interest.","order":2,"name":"Ethics","group":{"name":"EthicsHeading","label":"Conflict of Interest"}},{"value":"This article does not contain any studies involving animals performed and any studies involving human participants performed by any of the authors.","order":3,"name":"Ethics","group":{"name":"EthicsHeading","label":"Research Involving Human Participants and\/or Animals"}},{"value":"In this article, there is no human or animal sample was involved in this study.","order":4,"name":"Ethics","group":{"name":"EthicsHeading","label":"Informed Consent"}}],"article-number":"252"}}