{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,2,10]],"date-time":"2026-02-10T03:46:00Z","timestamp":1770695160048,"version":"3.49.0"},"publisher-location":"Singapore","reference-count":30,"publisher":"Springer Nature Singapore","isbn-type":[{"value":"9789819569564","type":"print"},{"value":"9789819569571","type":"electronic"}],"license":[{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"tdm","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"},{"start":{"date-parts":[[2026,1,1]],"date-time":"2026-01-01T00:00:00Z","timestamp":1767225600000},"content-version":"vor","delay-in-days":0,"URL":"https:\/\/www.springernature.com\/gp\/researchers\/text-and-data-mining"}],"content-domain":{"domain":["link.springer.com"],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2026]]},"DOI":"10.1007\/978-981-95-6957-1_43","type":"book-chapter","created":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T10:43:44Z","timestamp":1770633824000},"page":"602-615","update-policy":"https:\/\/doi.org\/10.1007\/springer_crossmark_policy","source":"Crossref","is-referenced-by-count":0,"title":["SE-EEND: A Structurally Enhanced End-to-End Neural Diarization System"],"prefix":"10.1007","author":[{"ORCID":"https:\/\/orcid.org\/0009-0004-5267-5111","authenticated-orcid":false,"given":"Penghao","family":"Ma","sequence":"first","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0000-0002-5461-8901","authenticated-orcid":false,"given":"Guangcun","family":"Wei","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0002-1926-5116","authenticated-orcid":false,"given":"Chuike","family":"Kong","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0007-8684-2450","authenticated-orcid":false,"given":"Shuo","family":"Li","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]},{"ORCID":"https:\/\/orcid.org\/0009-0000-2618-9313","authenticated-orcid":false,"given":"Jianfeng","family":"Fang","sequence":"additional","affiliation":[],"role":[{"role":"author","vocabulary":"crossref"}]}],"member":"297","published-online":{"date-parts":[[2026,2,10]]},"reference":[{"key":"43_CR1","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2021.101317","volume":"72","author":"TJ Park","year":"2022","unstructured":"Park, T.J., Kanda, N., Dimitriadis, D., Han, K.J., Watanabe, S., Narayanan, S.: A review of speaker diarization: recent advances with deep learning. Comput. Speech Lang. 72, 101317 (2022)","journal-title":"Comput. Speech Lang."},{"key":"43_CR2","doi-asserted-by":"publisher","unstructured":"Kanda, N., et al.: Joint speaker counting, speech recognition, and speaker identification for overlapped speech of any number of speakers. In: Interspeech 2020, pp. 36\u201340 (2020). https:\/\/doi.org\/10.21437\/Interspeech.2020-1085","DOI":"10.21437\/Interspeech.2020-1085"},{"key":"43_CR3","doi-asserted-by":"publisher","unstructured":"Lin, Q., Yin, R., Li, M., Bredin, H., Barras, C.: LSTM based similarity measurement with spectral clustering for speaker diarization. In: Interspeech 2019, pp. 366\u2013370 (2019). https:\/\/doi.org\/10.21437\/Interspeech.2019-1388","DOI":"10.21437\/Interspeech.2019-1388"},{"key":"43_CR4","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2021.101254","volume":"71","author":"F Landini","year":"2022","unstructured":"Landini, F., Profant, J., Diez, M., Burget, L.: Bayesian hmm clustering of x-vector sequences (vbx) in speaker diarization: theory, implementation and analysis on standard tasks. Comput. Speech Lang. 71, 101254 (2022)","journal-title":"Comput. Speech Lang."},{"key":"43_CR5","doi-asserted-by":"crossref","unstructured":"Jung, J.w., et al.: In search of strong embedding extractors for speaker diarisation. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10096449"},{"key":"43_CR6","doi-asserted-by":"crossref","unstructured":"Heo, H.S., Kwon, Y., Lee, B.J., Kim, Y.J., Jung, J.w.: High-resolution embedding extractor for speaker diarisation. In: ICASSP 2023-2023 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp.\u00a01\u20135. IEEE (2023)","DOI":"10.1109\/ICASSP49357.2023.10097190"},{"key":"43_CR7","doi-asserted-by":"publisher","unstructured":"Fujita, Y., Kanda, N., Horiguchi, S., Nagamatsu, K., Watanabe, S.: End-to-end neural speaker diarization with permutation-free objectives. In: Interspeech 2019, pp. 4300\u20134304 (2019). https:\/\/doi.org\/10.21437\/Interspeech.2019-2899","DOI":"10.21437\/Interspeech.2019-2899"},{"key":"43_CR8","doi-asserted-by":"crossref","unstructured":"Fujita, Y., Kanda, N., Horiguchi, S., Xue, Y., Nagamatsu, K., Watanabe, S.: End-to-end neural speaker diarization with self-attention. In: 2019 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp. 296\u2013303. IEEE (2019)","DOI":"10.1109\/ASRU46091.2019.9003959"},{"key":"43_CR9","doi-asserted-by":"publisher","unstructured":"Wang, C., Li, J., Fang, X., Kang, J., Li, Y.: End-to-end neural speaker diarization with absolute speaker loss. In: Interspeech 2023, pp. 3577\u20133581 (2023). https:\/\/doi.org\/10.21437\/Interspeech.2023-656","DOI":"10.21437\/Interspeech.2023-656"},{"key":"43_CR10","doi-asserted-by":"crossref","unstructured":"Fung, I., Samarakoon, L., Broughton, S.J.: Robust end-to-end diarization with domain adaptive training and multi-task learning. In: 2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.\u00a01\u20137. IEEE (2023)","DOI":"10.1109\/ASRU57964.2023.10389622"},{"key":"43_CR11","doi-asserted-by":"crossref","unstructured":"Yu, Y., Park, D., Kim, H.K.: Auxiliary loss of transformer with residual connection for end-to-end speaker diarization. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 8377\u20138381. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746602"},{"key":"43_CR12","doi-asserted-by":"crossref","unstructured":"Zeghidour, N., Teboul, O., Grangier, D.: Dive: end-to-end speech diarization via iterative speaker embedding. In: 2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp. 702\u2013709. IEEE (2021)","DOI":"10.1109\/ASRU51503.2021.9688178"},{"key":"43_CR13","doi-asserted-by":"crossref","unstructured":"Samarakoon, L., Broughton, S.J., H\u00e4rk\u00f6nen, M., Fung, I.: Transformer attractors for robust and efficient end-to-end neural diarization. In: 2023 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp.\u00a01\u20138. IEEE (2023)","DOI":"10.1109\/ASRU57964.2023.10389679"},{"key":"43_CR14","doi-asserted-by":"publisher","unstructured":"Chen, Z., Han, B., Wang, S., Qian, Y.: Attention-based encoder-decoder network for end-to-end neural speaker diarization with target speaker attractor. In: Interspeech 2023, pp. 3552\u20133556 (2023). https:\/\/doi.org\/10.21437\/Interspeech.2023-1228","DOI":"10.21437\/Interspeech.2023-1228"},{"key":"43_CR15","doi-asserted-by":"publisher","first-page":"566","DOI":"10.1016\/j.neunet.2023.07.043","volume":"166","author":"F Hao","year":"2023","unstructured":"Hao, F., Li, X., Zheng, C.: End-to-end neural speaker diarization with an iterative adaptive attractor estimation. Neural Netw. 166, 566\u2013578 (2023)","journal-title":"Neural Netw."},{"key":"43_CR16","doi-asserted-by":"crossref","unstructured":"Rybicka, M., Villalba, J., Dehak, N., Kowalczyk, K.: End-to-end neural speaker diarization with an iterative refinement of non-autoregressive attention-based attractors. In: INTERSPEECH, pp. 5090\u20135094 (2022)","DOI":"10.21437\/Interspeech.2022-10169"},{"key":"43_CR17","doi-asserted-by":"crossref","unstructured":"Horiguchi, S., Watanabe, S., Garc\u00eda, P., Xue, Y., Takashima, Y., Kawaguchi, Y.: Towards neural diarization for unlimited numbers of speakers using global and local attractors. In: 2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU), pp. 98\u2013105. IEEE (2021)","DOI":"10.1109\/ASRU51503.2021.9687875"},{"key":"43_CR18","doi-asserted-by":"crossref","unstructured":"Maiti, S., Erdogan, H., Wilson, K., Wisdom, S., Watanabe, S., Hershey, J.R.: End-to-end diarization for variable number of speakers with local-global networks and discriminative speaker embeddings. In: ICASSP 2021-2021 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 7183\u20137187. IEEE (2021)","DOI":"10.1109\/ICASSP39728.2021.9414841"},{"key":"43_CR19","doi-asserted-by":"publisher","unstructured":"Liu, Y.C., Han, E., Lee, C., Stolcke, A.: End-to-end neural diarization: From transformer to conformer. In: Interspeech 2021, pp. 3081\u20133085 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-1909","DOI":"10.21437\/Interspeech.2021-1909"},{"key":"43_CR20","doi-asserted-by":"crossref","unstructured":"Yu, D., Kolb\u00e6k, M., Tan, Z.H., Jensen, J.: Permutation invariant training of deep models for speaker-independent multi-talker speech separation. In: 2017 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 241\u2013245. IEEE (2017)","DOI":"10.1109\/ICASSP.2017.7952154"},{"key":"43_CR21","doi-asserted-by":"publisher","unstructured":"Prabhu, D., Peng, Y., Jyothi, P., Watanabe, S.: Multi-convformer: extending conformer with multiple convolution kernels. In: Interspeech 2024, pp. 232\u2013236 (2024). https:\/\/doi.org\/10.21437\/Interspeech.2024-2384","DOI":"10.21437\/Interspeech.2024-2384"},{"key":"43_CR22","doi-asserted-by":"publisher","unstructured":"Fu, Y., et al.: Aishell-4: An open source dataset for speech enhancement, separation, recognition and speaker diarization in conference scenario. In: Interspeech 2021, pp. 3665\u20133669 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-1397","DOI":"10.21437\/Interspeech.2021-1397"},{"key":"43_CR23","doi-asserted-by":"crossref","unstructured":"Yu, F., et\u00a0al.: M2met: The icassp 2022 multi-channel multi-party meeting transcription challenge. In: ICASSP 2022-2022 IEEE International Conference on Acoustics, Speech and Signal Processing (ICASSP), pp. 6167\u20136171. IEEE (2022)","DOI":"10.1109\/ICASSP43922.2022.9746465"},{"key":"43_CR24","unstructured":"Kraaij, W., Hain, T., Lincoln, M., Post, W.: The AMI meeting corpus. In: Proc. International Conference on Methods and Techniques in Behavioral Research, pp.\u00a01\u20134 (2005)"},{"key":"43_CR25","doi-asserted-by":"publisher","unstructured":"Ryant, N., et al.: The third dihard diarization challenge. In: Interspeech 2021, pp. 3570\u20133574 (2021). https:\/\/doi.org\/10.21437\/Interspeech.2021-1208","DOI":"10.21437\/Interspeech.2021-1208"},{"key":"43_CR26","doi-asserted-by":"publisher","unstructured":"Yang, Z., et al.: Open source magicdata-ramc: A rich annotated mandarin conversational (RAMC) speech dataset. In: Interspeech 2022, pp. 1736\u20131740 (2022). https:\/\/doi.org\/10.21437\/Interspeech.2022-729","DOI":"10.21437\/Interspeech.2022-729"},{"key":"43_CR27","unstructured":"Vaswani, A., et al.: Attention is all you need. Adv. Neural Inf. Process. Syst. 30 (2017)"},{"key":"43_CR28","doi-asserted-by":"crossref","unstructured":"Landini, F., Stafylakis, T., Burget, L., et al.: Diaper: end-to-end neural diarization with perceiver-based attractors. IEEE\/ACM Trans. Audio Speech Language Process. (2024)","DOI":"10.1109\/TASLP.2024.3422818"},{"key":"43_CR29","doi-asserted-by":"publisher","first-page":"1493","DOI":"10.1109\/TASLP.2022.3162080","volume":"30","author":"S Horiguchi","year":"2022","unstructured":"Horiguchi, S., Fujita, Y., Watanabe, S., Xue, Y., Garcia, P.: Encoder-decoder based attractors for end-to-end neural diarization. IEEE\/ACM Trans. Audio Speech Language Process. 30, 1493\u20131507 (2022)","journal-title":"IEEE\/ACM Trans. Audio Speech Language Process."},{"key":"43_CR30","doi-asserted-by":"publisher","unstructured":"Plaquet, A., Bredin, H.: Powerset multi-class cross entropy loss for neural speaker diarization. In: Interspeech 2023, pp. 3222\u20133226 (2023). https:\/\/doi.org\/10.21437\/Interspeech.2023-205","DOI":"10.21437\/Interspeech.2023-205"}],"container-title":["Lecture Notes in Computer Science","MultiMedia Modeling"],"original-title":[],"language":"en","link":[{"URL":"https:\/\/link.springer.com\/content\/pdf\/10.1007\/978-981-95-6957-1_43","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2026,2,9]],"date-time":"2026-02-09T10:43:50Z","timestamp":1770633830000},"score":1,"resource":{"primary":{"URL":"https:\/\/link.springer.com\/10.1007\/978-981-95-6957-1_43"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2026]]},"ISBN":["9789819569564","9789819569571"],"references-count":30,"URL":"https:\/\/doi.org\/10.1007\/978-981-95-6957-1_43","relation":{},"ISSN":["0302-9743","1611-3349"],"issn-type":[{"value":"0302-9743","type":"print"},{"value":"1611-3349","type":"electronic"}],"subject":[],"published":{"date-parts":[[2026]]},"assertion":[{"value":"10 February 2026","order":1,"name":"first_online","label":"First Online","group":{"name":"ChapterHistory","label":"Chapter History"}},{"value":"MMM","order":1,"name":"conference_acronym","label":"Conference Acronym","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"International Conference on Multimedia Modeling","order":2,"name":"conference_name","label":"Conference Name","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Prague","order":3,"name":"conference_city","label":"Conference City","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"Czech Republic","order":4,"name":"conference_country","label":"Conference Country","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"2026","order":5,"name":"conference_year","label":"Conference Year","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"29 January 2026","order":7,"name":"conference_start_date","label":"Conference Start Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"31 January 2026","order":8,"name":"conference_end_date","label":"Conference End Date","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"32","order":9,"name":"conference_number","label":"Conference Number","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"mmm2026","order":10,"name":"conference_id","label":"Conference ID","group":{"name":"ConferenceInfo","label":"Conference Information"}},{"value":"https:\/\/mmm2026.cz\/","order":11,"name":"conference_url","label":"Conference URL","group":{"name":"ConferenceInfo","label":"Conference Information"}}]}}