{"status":"ok","message-type":"work","message-version":"1.0.0","message":{"indexed":{"date-parts":[[2026,7,23]],"date-time":"2026-07-23T15:35:09Z","timestamp":1784820909503,"version":"3.55.0"},"reference-count":31,"publisher":"IEEE","license":[{"start":{"date-parts":[[2021,12,13]],"date-time":"2021-12-13T00:00:00Z","timestamp":1639353600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-029"},{"start":{"date-parts":[[2021,12,13]],"date-time":"2021-12-13T00:00:00Z","timestamp":1639353600000},"content-version":"stm-asf","delay-in-days":0,"URL":"https:\/\/doi.org\/10.15223\/policy-037"}],"funder":[{"DOI":"10.13039\/501100010190","name":"GENCI","doi-asserted-by":"publisher","award":["AD011012177"],"award-info":[{"award-number":["AD011012177"]}],"id":[{"id":"10.13039\/501100010190","id-type":"DOI","asserted-by":"publisher"}]},{"DOI":"10.13039\/501100001665","name":"French National Research Agency (ANR)","doi-asserted-by":"publisher","award":["ANR-16-CE92-0025"],"award-info":[{"award-number":["ANR-16-CE92-0025"]}],"id":[{"id":"10.13039\/501100001665","id-type":"DOI","asserted-by":"publisher"}]}],"content-domain":{"domain":[],"crossmark-restriction":false},"short-container-title":[],"published-print":{"date-parts":[[2021,12,13]]},"DOI":"10.1109\/asru51503.2021.9688044","type":"proceedings-article","created":{"date-parts":[[2022,2,3]],"date-time":"2022-02-03T20:31:00Z","timestamp":1643920260000},"page":"1139-1146","source":"Crossref","is-referenced-by-count":26,"title":["Overlap-Aware Low-Latency Online Speaker Diarization Based on End-to-End Local Segmentation"],"prefix":"10.1109","author":[{"given":"Juan M.","family":"Coria","sequence":"first","affiliation":[{"name":"Universit&#x00E9; Paris-Saclay CNRS, LISN,Orsay,France"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Herve","family":"Bredin","sequence":"additional","affiliation":[{"name":"IRIT, Universit&#x00E9; de Toulouse, CNRS,Toulouse,France"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sahar","family":"Ghannay","sequence":"additional","affiliation":[{"name":"Universit&#x00E9; Paris-Saclay CNRS, LISN,Orsay,France"}],"role":[{"vocabulary":"crossref","role":"author"}]},{"given":"Sophie","family":"Rosset","sequence":"additional","affiliation":[{"name":"Universit&#x00E9; Paris-Saclay CNRS, LISN,Orsay,France"}],"role":[{"vocabulary":"crossref","role":"author"}]}],"member":"263","reference":[{"key":"ref31","first-page":"3587","article-title":"pyannote.metrics: A Toolkit for Re-producible Evaluation, Diagnostic, and Error Analysis of Speaker Diarization Systems","author":"bredin","year":"0","journal-title":"Proc Interspeech 2017"},{"key":"ref30","article-title":"Algorithms for Hyper-Parameter Op-timization","volume":"24","author":"bergstra","year":"2011","journal-title":"Advances in Neural Information Pro-cessing Systems"},{"key":"ref10","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2021-560"},{"key":"ref11","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-2899"},{"key":"ref12","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU46091.2019.9003959"},{"key":"ref13","first-page":"269","article-title":"End-to-End Speaker Diarization for an Unknown Number of Speak-ers with Encoder-Decoder Based Attractors","author":"horiguchi","year":"0","journal-title":"Proc Interspeech 2020"},{"key":"ref14","article-title":"Auto-Tuning Spectral Clustering for Speaker Diarization Using Normalized Maxi-mum Eigengap","author":"park","year":"2019","journal-title":"IEEE Signal Processing Letters"},{"key":"ref15","first-page":"7198","article-title":"Integrating end-to-end neural and clustering-based di-arization: Getting the best of both worlds","author":"kinoshita","year":"0","journal-title":"ICASSP 2021 &#x2014; 2021 IEEE International Conference on Acous-tics Speech and Signal Processing (ICASSP)"},{"key":"ref16","doi-asserted-by":"publisher","DOI":"10.1016\/j.csl.2021.101254"},{"key":"ref17","article-title":"Online Streaming End-to-End Neu-ral Diarization Handling Overlapping Speech and Flexible Numbers of Speakers","author":"xue","year":"2021","journal-title":"ArXiv Preprint"},{"key":"ref18","article-title":"The Third DIHARD Diarization Challenge","author":"ryant","year":"2020","journal-title":"ArXiv Preprint"},{"key":"ref19","doi-asserted-by":"publisher","DOI":"10.1007\/s10579-007-9040-x"},{"key":"ref28","doi-asserted-by":"crossref","DOI":"10.1073\/pnas.1612524113","article-title":"Statistics of nat-ural reverberation enable perceptual separation of sound and space","volume":"113","author":"traer","year":"2016","journal-title":"Proceedings of the National Academy of Sciences"},{"key":"ref4","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP.2018.8461375"},{"key":"ref27","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2018-1929"},{"key":"ref3","article-title":"Integrating online i-vector extractor with information bottleneck based speaker diarization sys-tem","author":"madikeri","year":"0","journal-title":"Proc Interspeech 2015"},{"key":"ref6","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9054251"},{"key":"ref29","article-title":"MU-SAN: A Music, Speech, and Noise Corpus","author":"snyder","year":"2015","journal-title":"ArXiv Preprint"},{"key":"ref5","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2019-1388"},{"key":"ref8","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP40776.2020.9053096"},{"key":"ref7","doi-asserted-by":"publisher","DOI":"10.1109\/ASRU.2007.4430194"},{"key":"ref2","first-page":"2798","article-title":"BUT Sys-tem for DIHARD Speech Diarization Challenge 2018","author":"diez","year":"0","journal-title":"Proc Interspeech 2018"},{"key":"ref9","doi-asserted-by":"publisher","DOI":"10.1109\/ICASSP39728.2021.9413436"},{"key":"ref1","doi-asserted-by":"publisher","DOI":"10.1109\/TASL.2011.2125954"},{"key":"ref20","doi-asserted-by":"publisher","DOI":"10.21437\/Interspeech.2020-2337"},{"key":"ref22","article-title":"The Third DIHARD Diarization Challenge","author":"ryant","year":"2020","journal-title":"ArXiv Preprint"},{"key":"ref21","first-page":"978","article-title":"The Second DIHARD Diarization Chal-lenge: Dataset, Task, and Baselines","author":"ryant","year":"0","journal-title":"Proc INTER-SPEECH 2019"},{"key":"ref24","doi-asserted-by":"publisher","DOI":"10.1109\/SLT.2018.8639585"},{"key":"ref23","article-title":"Third DIHARD Challenge Evaluation Plan","author":"ryant","year":"2020","journal-title":"ArXiv Preprint"},{"key":"ref26","first-page":"2616","article-title":"VoxCeleb: A Large-Scale Speaker Identification Dataset","author":"nagrani","year":"0","journal-title":"Proc Interspeech 2017"},{"key":"ref25","first-page":"4685","article-title":"Ar-cFace: Additive Angular Margin Loss for Deep Face Recognition","author":"deng","year":"0","journal-title":"2019 IEEE Conference on Computer Vision and Pattern Recognition (CVPR)"}],"event":{"name":"2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)","location":"Cartagena, Colombia","start":{"date-parts":[[2021,12,13]]},"end":{"date-parts":[[2021,12,17]]}},"container-title":["2021 IEEE Automatic Speech Recognition and Understanding Workshop (ASRU)"],"original-title":[],"link":[{"URL":"http:\/\/xplorestaging.ieee.org\/ielx7\/9687821\/9687855\/09688044.pdf?arnumber=9688044","content-type":"unspecified","content-version":"vor","intended-application":"similarity-checking"}],"deposited":{"date-parts":[[2022,5,16]],"date-time":"2022-05-16T20:41:56Z","timestamp":1652733716000},"score":1,"resource":{"primary":{"URL":"https:\/\/ieeexplore.ieee.org\/document\/9688044\/"}},"subtitle":[],"short-title":[],"issued":{"date-parts":[[2021,12,13]]},"references-count":31,"URL":"https:\/\/doi.org\/10.1109\/asru51503.2021.9688044","relation":{},"subject":[],"published":{"date-parts":[[2021,12,13]]}}}